reorg: move system/devices into system/kernel/devices

With the shared device data (device-abi, pci-class, usb-abi, usb-ids, acpi-ids,
aml) now in library/device/, what remained in system/devices/ was purely
kernel-internal: the firmware-discovery machinery and the rich pointer-based
device model (platform, device-model, acpi, device-tree, power), reached only
through the "platform" module by three kernel files. It belongs with the kernel.

Move it to system/kernel/devices/. A pure relocation: the "platform" module
name is unchanged and every cross-dir dependency is a module import, so only the
one b.path and some comments move. The top-level split is now clean —
system/kernel/ is the kernel, library/device/ the shared device libraries,
system/{drivers,services} the userspace. The runtime /system/devices concept
(the virtual device tree) is unaffected; system/kernel/devices/ is its
implementation.

Also fixes two comment refs that still pointed device-abi at its pre-Wave-1a
home (system/devices/) — it lives at library/device/model/ now.

zig build + test green; smoke, discovery, acpi-parse, acpi-ps2, acpi-report pass.
This commit is contained in:
Daniel Samson
2026-07-22 21:35:30 +01:00
parent 5b874fc756
commit 3a155cdc7d
13 changed files with 15 additions and 14 deletions
+841
View File
@@ -0,0 +1,841 @@
//! ACPI discovery backend.
//!
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
//! bootloader handed us) and translates the static tables into the generic
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
//! source. This is deliberately the *static-table* path, and **only** that: MADT
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
//! interpreted here — the kernel collects the blobs and publishes them on the
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
//! off the single-core critical path (nothing else runs alongside it there).
//!
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
//! and is *not* mapped up front, so configuration-space pages are mapped on demand via
//! the `Hal.mapMmio` callback the caller supplies (the architecture VMM's map primitive).
const std = @import("std");
const boot_handoff = @import("boot-handoff");
const abi = @import("abi");
const parameters = @import("parameters");
const device_model = @import("device-model.zig");
const DeviceTree = device_model.DeviceTree;
const Hal = device_model.Hal;
/// A hardware register located either in MMIO or I/O-port space, as ACPI's
/// Generic Address Structure describes. `address == 0` means "not present".
pub const RegisterAccess = struct {
/// true = system memory (MMIO), false = system I/O port space.
mmio: bool = false,
address: u64 = 0,
/// Access width in bytes.
width: u8 = 0,
pub fn present(self: RegisterAccess) bool {
return self.address != 0;
}
};
/// The power register map, extracted from the FADT during discovery. Populated by
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
/// writes the PM1 control register itself). So this holds only the FADT scalars.
pub const PowerInformation = struct {
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
sci_interrupt: u16 = 0,
/// The SMM command port and the value that switches the platform into ACPI mode.
smi_cmd: u16 = 0,
acpi_enable: u8 = 0,
acpi_disable: u8 = 0,
/// PM1 control registers — writing SLP_TYP|SLP_EN here enters a sleep state.
pm1a_cnt: RegisterAccess = .{},
pm1b_cnt: RegisterAccess = .{},
/// The FADT reset register and the value to write to it.
reset: RegisterAccess = .{},
reset_value: u8 = 0,
reset_supported: bool = false,
};
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
pub var power_information: PowerInformation = .{};
/// A legacy ISA IRQ remapped to a different global system interrupt (GSI), from a
/// MADT Interrupt Source Override. `flags` are the MPS INTI polarity/trigger bits.
pub const IsoEntry = struct {
source: u8,
gsi: u32,
flags: u16,
};
/// Firmware facts the architecture layer needs to avoid legacy assumptions (so danos boots
/// on legacy-free UEFI Class 3 machines). MMIO device *addresses* (HPET, IOAPIC)
/// come from the device tree instead; this holds the scalar facts that have no
/// natural device node.
pub const PlatformInformation = struct {
/// Whether the legacy 8259 PIC is present (MADT flags bit 0, PCAT_COMPAT). When
/// false, the PIC must not be programmed (it may not exist).
pic_present: bool = false,
/// Local APIC MMIO base (MADT, honouring a type-5 address override).
lapic_base: u64 = 0xFEE00000,
/// The ACPI power-management timer — a fixed 3.579545 MHz counter usable as a
/// calibration reference when no HPET is present.
pm_timer: RegisterAccess = .{},
/// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VALUE_EXT).
pm_timer_32bit: bool = false,
/// The console UART the firmware points at (SPCR), if any — MMIO or I/O port.
spcr_uart: ?RegisterAccess = null,
/// SPCR interface type (0/1 = 16550/16450, …).
spcr_kind: u8 = 0,
/// ISA-IRQ-to-GSI remappings from the MADT (for future IOAPIC routing).
overrides: [16]IsoEntry = undefined,
override_count: usize = 0,
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
/// Detection is the first step; per-device domain enforcement lands with the first
/// DMA driver.
iommu_present: bool = false,
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
iommu_base: u64 = 0,
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
/// reading it back nonzero confirms a real, mappable VT-d unit.
iommu_version: u32 = 0,
/// The unit's Capability register (offset 0x08): supported address widths, number
/// of domains, etc. Recorded now; consumed when enforcement is built.
iommu_capabilities: u64 = 0,
};
/// Filled in by `discover`; the architecture layer reads it during bring-up.
pub var platform_information: PlatformInformation = .{};
/// One usable logical processor, from a MADT type-0 (Local APIC) record. The
/// `apic_id` is the Local APIC ID that SMP bring-up targets to wake this core
/// (INIT–SIPI–SIPI); `processor_id` is the ACPI namespace handle. Only processors
/// the firmware marks *enabled* are recorded — a disabled one can't be started.
pub const Cpu = struct {
processor_id: u8,
apic_id: u8,
/// MADT flags bit 1: usable but firmware-started offline (hot-plug / deferred
/// bring-up), as opposed to already available. Informational for now.
online_capable: bool,
};
/// The set of usable logical processors the MADT listed — the hardware's degree of
/// parallelism. Includes the bootstrap processor danos already runs on; the rest
/// are the application processors SMP bring-up would start (see docs/smp.md).
pub const CpuInformation = struct {
/// A static pool sized well above any danos target (a desktop, two 4-core Pis).
/// If the MADT ever lists more, the surplus is dropped and counted in `dropped`
/// so the truncation is never silent.
cpus: [maximum_cpus]Cpu = undefined,
count: usize = 0,
dropped: usize = 0,
};
const maximum_cpus = parameters.maximum_cpus;
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
pub var cpu_information: CpuInformation = .{};
/// Physical address of the DSDT the FADT points at, or 0.
pub var dsdt_physical: u64 = 0;
/// The FADT itself (physical + length), published on the acpi-tables node so
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
var fadt_physical: u64 = 0;
var fadt_length: u64 = 0;
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
// address + length of each table's post-header bytecode. The kernel does not
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
var aml_block_physical: [32]u64 = undefined;
var aml_block_len: [32]usize = undefined;
var aml_block_count: usize = 0;
fn addAmlBlock(sdt_physical: u64) void {
if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return;
const h: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
aml_block_physical[aml_block_count] = sdt_physical + @sizeOf(SystemDescriptorTableHeader);
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
aml_block_count += 1;
}
/// RSDP structure for revision 0 (version 1.0)
const RootSystemDescriptionPointer = extern struct {
/// An 8 byte magic number used for locating the RSDP, containing RSD PTR.
signature: [8]u8,
/// A byte used to verify the first 20 bytes of the RSDP
checksum: u8,
/// An OEM-supplied string that identified the OEM.
oem_id: [6]u8,
/// The RSDP revision, used for determining which fields are available.
revision: u8,
/// A 32-bit physical address pointing to the RSDT.
root_system_description_table_address: u32 align(1),
};
/// XSDP structure for revision 2 (version 2.0+)
const ExtendedSystemDescriptorPointer = extern struct {
/// An 8 byte magic number used for locating the RSDP, containing RSD PTR.
signature: [8]u8,
/// A byte used to verify the first 20 bytes of the RSDP
checksum: u8,
/// An OEM-supplied string that identified the OEM.
oem_id: [6]u8,
/// The RSDP revision, used for determining which fields are available.
revision: u8,
/// deprecated since version 2.0. A 32-bit physical address pointing to the RSDT.
root_system_description_table_address: u32 align(1),
/// The size of the RSDP.
length: u32 align(1),
/// A 64-bit physical address pointing to the XSDT. If the revision is at least 2, the XSDT
/// should be used regardless of architecture, as the RSDT was deprecated.
extended_system_descriptor_table_address: u64 align(1),
/// A checksum used for the entire table.
extended_checksum: u8,
reserved: [3]u8,
};
/// Multiple APIC Description Table (MADT)
const APIC: [4]u8 = "APIC".*;
/// Boot Error Record Table (BERT)
const BERT: [4]u8 = "BERT".*;
/// Corrected Platform Error Polling Table (CPEP)
const CPEP: [4]u8 = "CPEP".*;
/// Differentiated System Description Table (DSDT)
const DSDT: [4]u8 = "DSDT".*;
/// Embedded Controller Boot Resources Table (ECDT)
const ECDT: [4]u8 = "ECDT".*;
/// Error Injection Table (EINJ)
const EINJ: [4]u8 = "EINJ".*;
/// Error Record Serialization Table (ERST)
const ERST: [4]u8 = "ERST".*;
/// Fixed ACPI Description Table (FADT)
const FACP: [4]u8 = "FACP".*;
/// Firmware ACPI Control Structure (FACS)
const FACS: [4]u8 = "FACS".*;
/// Hardware Error Source Table (HEST)
const HEST: [4]u8 = "HEST".*;
/// High Precision Event Timer table (HPET)
const HPET: [4]u8 = "HPET".*;
/// PCI Express memory-mapped configuration space table (MCFG)
const MCFG: [4]u8 = "MCFG".*;
/// Maximum System Characteristics Table (MSCT)
const MSCT: [4]u8 = "MSCT".*;
/// Memory Power State Table (MPST)
const MPST: [4]u8 = "MPST".*;
// Platform Memory Topology Table (PMTT)
const PMTT: [4]u8 = "PMTT".*;
/// Persistent System Description Table (PSDT)
const PSDT: [4]u8 = "PSDT".*;
/// ACPI RAS Feature Table (RASF)
const RASF: [4]u8 = "RASF".*;
/// Root System Description Table
const RSDT: [4]u8 = "RSDT".*;
/// Smart Battery Specification Table (SBST)
const SBST: [4]u8 = "SBST".*;
/// System Locality System Information Table (SLIT)
const SLIT: [4]u8 = "SLIT".*;
/// System Resource Affinity Table (SRAT)
const SRAT: [4]u8 = "SRAT".*;
/// Secondary System Description Table (SSDT)
const DMAR: [4]u8 = "DMAR".*;
const SSDT: [4]u8 = "SSDT".*;
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
const SPCR: [4]u8 = "SPCR".*;
/// Extended System Description Table (XSDT; 64-bit version of the RSDT)
const XSDT: [4]u8 = "XSDT".*;
/// The header every system descriptor table (RSDT/XSDT and each SDT) begins with.
const SystemDescriptorTableHeader = extern struct {
/// A 4 byte signature used for identification (e.g. "RSDT", "APIC").
signature: [4]u8,
/// The length of the entire table, including the header.
length: u32 align(1),
/// The revision of the ACPI spec this table conforms to.
revision: u8,
/// An 8-bit checksum field for the whole table, inclusive of the header.
checksum: u8,
/// An OEM-supplied string that identified the OEM.
oem_id: [6]u8,
oem_table_id: [8]u8,
oem_revision: u32 align(1),
creator_id: u32 align(1),
creator_revision: u32 align(1),
};
// --- MADT: Multiple APIC Description Table (signature "APIC") ---------------
const Madt = extern struct {
header: SystemDescriptorTableHeader,
local_apic_address: u32 align(1),
flags: u32 align(1),
// Followed by a variable-length run of interrupt-controller records, each a
// MadtRecordHeader plus a type-specific body.
};
const MadtRecordHeader = extern struct {
type: u8,
length: u8,
};
/// MADT record type 0: a processor's Local APIC.
const MadtLocalApic = extern struct {
record: MadtRecordHeader,
processor_id: u8,
apic_id: u8,
/// bit 0 = enabled, bit 1 = online-capable.
flags: u32 align(1),
};
/// MADT record type 1: an I/O APIC.
const MadtIoApic = extern struct {
record: MadtRecordHeader,
io_apic_id: u8,
reserved: u8,
address: u32 align(1),
/// First global system interrupt this I/O APIC handles.
gsi_base: u32 align(1),
};
/// MADT record type 2: an Interrupt Source Override (ISA IRQ -> GSI remap).
const MadtIso = extern struct {
record: MadtRecordHeader,
bus: u8,
source: u8,
gsi: u32 align(1),
flags: u16 align(1),
};
/// MADT record type 5: Local APIC Address Override (64-bit MMIO base).
const MadtLapicOverride = extern struct {
record: MadtRecordHeader,
reserved: u16 align(1),
address: u64 align(1),
};
// --- MCFG: PCIe ECAM configuration space (signature "MCFG") -----------------
const Mcfg = extern struct {
header: SystemDescriptorTableHeader,
reserved: u64 align(1),
// Followed by one or more McfgAllocation entries.
};
const McfgAllocation = extern struct {
/// Physical base of this segment group's ECAM window.
base_address: u64 align(1),
segment_group: u16 align(1),
start_bus: u8,
end_bus: u8,
reserved: u32 align(1),
};
// --- HPET (signature "HPET") ------------------------------------------------
const Hpet = extern struct {
header: SystemDescriptorTableHeader,
hardware_rev_id: u8,
flags: u8,
pci_vendor_id: u16 align(1),
// Generic Address Structure describing the register block.
address_space_id: u8,
register_bit_width: u8,
register_bit_offset: u8,
gas_reserved: u8,
address: u64 align(1),
hpet_number: u8,
minimum_tick: u16 align(1),
page_protection: u8,
};
// --- Entry point ------------------------------------------------------------
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
if (rsdp_physical == 0) return error.NoRsdp;
boot_memory_regions = memory_regions;
// Start clean so a re-run doesn't accumulate stale state.
power_information = .{};
fadt_physical = 0;
fadt_length = 0;
platform_information = .{};
dsdt_physical = 0;
aml_block_count = 0;
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum;
if (rsdp.revision >= 2) {
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum;
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, device_tree, hal);
} else {
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
}
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
// and the power register map. The AML bytecode (device enumeration and the
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
// job: it claims the acpi-tables node published below, parses the same blobs,
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
// AML in the kernel keeps boot latency off the critical, single-core path.
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
// memory resources for the acpi service to map and parse in ring 3, a broad
// io_port grant for the OperationRegion access its interpreter needs, and
// the SCI for the events track (M21). Exactly one node, one trusted
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
// now that the kernel keeps only the *static* tables for itself.
publishAcpiTablesNode(device_tree) catch {};
}
/// Build the acpi-tables node (see the call site in discover). Best-effort: a
/// failure here leaves the kernel-seeded tree working, only the ring-3 service
/// finds nothing to claim.
fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
const node = try device_tree.addChild(device_tree.root, .acpi_tables, "acpi-tables");
// One memory resource per AML block — page-aligned base down, length padded
// up to cover the bytecode, so mmio_map hands the service a pointer into it.
var i: usize = 0;
while (i < aml_block_count and i < device_model.maximum_resources - 2) : (i += 1) {
// mmio_map preserves the sub-page offset, so the service maps this and
// gets a pointer straight to the bytecode.
_ = node.addResource(.memory, aml_block_physical[i], aml_block_len[i]);
}
// The broad I/O grant: OperationRegions name whatever ports the firmware
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
// that names them is parsed, so the grant is the whole space — the honest
// trust boundary of docs/discovery.md (the acpi service's one trusted node).
_ = node.addResource(.io_port, 0, 1 << 16);
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
// 1 and 12, the RTC, …), and the service registers those devices under this
// node, so it must own a superset. The range [0, 256) covers every GSI; the
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
_ = node.addResource(.irq, 0, 256);
// The FADT rides along (M21): the service reads the PM1 event / GPE blocks
// from its own copy, telling it apart from the AML blobs by signature.
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
}
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
/// each SDT it points at. A bad individual table is skipped, not fatal.
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum;
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
const base: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
for (entries[0..count]) |ent| {
const sdt_physical: u64 = ent; // u32 entries widen; u64 pass through
handleTable(device_tree, hal, sdt_physical) catch continue;
}
}
/// Dispatch a single SDT on its signature.
fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
const sig = header.signature;
if (std.mem.eql(u8, &sig, &APIC)) {
try parseMadt(device_tree, header);
} else if (std.mem.eql(u8, &sig, &MCFG)) {
try parseMcfg(device_tree, header);
} else if (std.mem.eql(u8, &sig, &HPET)) {
try parseHpet(device_tree, hal, header);
} else if (std.mem.eql(u8, &sig, &FACP)) {
fadt_physical = sdt_physical;
fadt_length = header.length;
parseFadt(header);
} else if (std.mem.eql(u8, &sig, &SPCR)) {
parseSpcr(header);
} else if (std.mem.eql(u8, &sig, &DMAR)) {
parseDmar(hal, header);
} else if (std.mem.eql(u8, &sig, &SSDT)) {
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
addAmlBlock(sdt_physical);
}
// Any other signature is recognised but left opaque for now.
}
/// MADT -> one processor node per Local APIC, one interrupt_controller per I/O APIC.
fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
const madt: *const Madt = @ptrCast(header);
const total: usize = header.length;
const base: [*]const u8 = @ptrCast(header);
var ioapic_index: usize = 0;
// MADT header: local APIC base + flags (bit 0 = 8259 PIC present).
platform_information.lapic_base = madt.local_apic_address;
platform_information.pic_present = madt.flags & 1 != 0;
var off: usize = @sizeOf(Madt);
while (off + @sizeOf(MadtRecordHeader) <= total) {
const rec: *const MadtRecordHeader = @ptrCast(base + off);
if (rec.length < @sizeOf(MadtRecordHeader)) break; // malformed; avoid a spin
switch (rec.type) {
0 => {
const la: *const MadtLocalApic = @ptrCast(base + off);
// bit 0 = enabled: skip processors the firmware marks unusable.
if (la.flags & 1 != 0) {
var nb: [24]u8 = undefined;
const nm = std.fmt.bufPrint(&nb, "cpu{d}", .{la.processor_id}) catch "cpu";
_ = try device_tree.addChild(device_tree.root, .processor, nm);
// Also record it as a schedulable core (with the APIC ID an AP
// wake needs, which the device node name doesn't preserve).
if (cpu_information.count < cpu_information.cpus.len) {
cpu_information.cpus[cpu_information.count] = .{
.processor_id = la.processor_id,
.apic_id = la.apic_id,
.online_capable = la.flags & 2 != 0,
};
cpu_information.count += 1;
} else {
cpu_information.dropped += 1;
}
}
},
1 => {
const io: *const MadtIoApic = @ptrCast(base + off);
var nb: [24]u8 = undefined;
const nm = std.fmt.bufPrint(&nb, "ioapic{d}", .{ioapic_index}) catch "ioapic";
ioapic_index += 1;
const d = try device_tree.addChild(device_tree.root, .interrupt_controller, nm);
_ = d.addResource(.memory, io.address, 0x20);
// The GSI range this I/O APIC handles, starting at gsi_base.
_ = d.addResource(.irq, io.gsi_base, 0);
},
2 => {
const iso: *const MadtIso = @ptrCast(base + off);
if (platform_information.override_count < platform_information.overrides.len) {
platform_information.overrides[platform_information.override_count] = .{
.source = iso.source,
.gsi = iso.gsi,
.flags = iso.flags,
};
platform_information.override_count += 1;
}
},
5 => {
const ovr: *const MadtLapicOverride = @ptrCast(base + off);
platform_information.lapic_base = ovr.address;
},
else => {},
}
off += rec.length;
}
}
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
const total: usize = header.length;
const base: [*]const u8 = @ptrCast(header);
var off: usize = @sizeOf(Mcfg);
while (off + @sizeOf(McfgAllocation) <= total) : (off += @sizeOf(McfgAllocation)) {
const alloc: *const McfgAllocation = @ptrCast(base + off);
const bus_count: u64 = @as(u64, alloc.end_bus - alloc.start_bus) + 1;
var nb: [24]u8 = undefined;
const nm = std.fmt.bufPrint(&nb, "pci{d}", .{alloc.segment_group}) catch "pci";
const bridge = try device_tree.addChild(device_tree.root, .pci_host_bridge, nm);
// ECAM window: 1 MiB of configuration space per bus.
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
addBridgeApertures(bridge);
// The bridge decodes the whole 16-bit I/O space toward its bus — the
// window functions' I/O BARs must register-contain within (M19.2).
_ = bridge.addResource(.io_port, 0, 1 << 16);
// The function walk itself retired to ring 3 (M19.3): the pci-bus
// driver claims this bridge, repeats the scan through its ECAM grant,
// and device_registers what it finds — the kernel seeds only the
// bridge. The scan's equivalence was proven before the hand-off
// (pci-scan), and the walk's history is in git if archaeology calls.
}
}
/// The boot memory map, stored at discover() entry for the aperture derivation
/// below (and, in M20, for the acpi-tables node's containment windows).
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
/// The bridge's MMIO apertures, derived from the boot memory map's holes
/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR
/// resources, and `device_register` containment demands the bridge own windows
/// that cover them. Everything the firmware described is "not hole"; the low
/// aperture runs from the end of the described space below 4 GiB up to the
/// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to
/// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no
/// matter what later moved to user space.
fn addBridgeApertures(bridge: *device_model.Device) void {
// Below 4 GiB the described regions are sparse (RAM low, firmware flash
// and tables high), so the holes are the *gaps between* them — a single
// "after the last region" rule dies on OVMF's flash at the very top.
// Sort-merge the described ranges, then keep the three largest gaps
// (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the
// high aperture fits). Above 4 GiB one aperture runs from the end of the
// described space to the 46-bit line.
const Range = struct { base: u64, end: u64 };
var below: [64]Range = undefined;
var below_count: usize = 0;
var high_end: u64 = 1 << 32;
for (boot_memory_regions) |region| {
const end = region.base + region.pages * 4096;
// Above 4 GiB only *usable RAM* blocks the aperture: OVMF describes
// its own 64-bit PCI window as a reserved region and then programs
// BARs inside it — honoring reserved there would exclude the very
// space BARs live in. Below 4 GiB every described region blocks (the
// kernel image, the tables, the ramdisk all live there). Bring-up
// trust: only the bridge's claimant can register into the aperture.
if (region.kind == .usable and end > high_end) high_end = end;
if (region.base >= (1 << 32) or below_count == below.len) continue;
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
below_count += 1;
}
// Insertion sort by base (the map is small and this runs once at boot).
for (1..below_count) |i| {
const key = below[i];
var j = i;
while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1];
below[j] = key;
}
// Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB.
var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3;
var cursor: u64 = 0;
var index: usize = 0;
while (index <= below_count) : (index += 1) {
const gap_end = if (index == below_count) (1 << 32) else below[index].base;
if (gap_end > cursor and gap_end - cursor >= (1 << 20)) {
// Keep the three largest, replacing the smallest kept so far.
var smallest: usize = 0;
for (gaps, 0..) |gap, gi| {
if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi;
}
if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) {
gaps[smallest] = .{ .base = cursor, .end = gap_end };
}
}
if (index < below_count and below[index].end > cursor) cursor = below[index].end;
}
for (gaps) |gap| {
if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base);
}
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
}
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
/// its comparators can raise.
///
/// Unlike a PCI device or an ACPI `_CRS` node, the HPET table carries **no interrupt
/// number**: which I/O APIC inputs a comparator may drive is advertised at runtime,
/// as a bitmask in `Tn_INT_ROUTE_CAP` (bits 63:32 of the Timer 0 configuration register).
/// So discovery maps the register block, reads the mask, and records one concrete
/// `irq` resource — the GSI a driver is entitled to bind. The driver commits to it
/// by writing `Tn_INT_ROUTE_CNF`; the kernel checks the binding against this
/// resource (see process.ownedGsi), which is what keeps `irq_bind` a capability
/// rather than a request for an arbitrary interrupt line.
fn parseHpet(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
const hpet: *const Hpet = @ptrCast(header);
const d = try device_tree.addChild(device_tree.root, .timer, "hpet");
// The GAS tag must say System Memory (0) before we treat `address` as a physical
// address. The HPET spec mandates it, but firmware is not a thing to trust: a
// System I/O (1) tag here would have us map an arbitrary page and read a bogus
// route-capability mask out of it.
if (hpet.address_space_id != gas_system_memory) return;
_ = d.addResource(.memory, hpet.address, 0x400);
const regs = hal.mapMmio(hpet.address, 0x400, true);
const t0_configuration: *const volatile u64 = @ptrFromInt(regs + 0x100);
const route_cap: u32 = @truncate(t0_configuration.* >> 32);
if (hpetGsi(route_cap)) |gsi| _ = d.addResource(.irq, gsi, 1);
}
/// ACPI Generic Address Structure address-space ids we care about.
const gas_system_memory: u8 = 0;
/// Pick a GSI for the HPET out of its route-capability mask. Prefer an input at or
/// above 16: the low ones overlap the legacy ISA lines (2 = cascaded PIT, 8 = RTC),
/// which the MADT may separately override, whereas 16+ are the free upper inputs on
/// every I/O APIC we care about. Falls back to the lowest bit set if there are none.
fn hpetGsi(route_cap: u32) ?u32 {
if (route_cap == 0) return null;
var gsi: u32 = 16;
while (gsi < 32) : (gsi += 1) {
if (route_cap & (@as(u32, 1) << @intCast(gsi)) != 0) return gsi;
}
return @ctz(route_cap);
}
// FADT field offsets (bytes from the table start). The FADT grew across ACPI
// revisions, so every field is read through `fadt()` with a length guard rather
// than a fixed struct — an older/shorter FADT simply lacks the later (X_) fields.
const fadt_dsdt = 40; // u32
const fadt_smi_cmd = 48; // u32 (an I/O port)
const fadt_acpi_enable = 52; // u8
const fadt_acpi_disable = 53; // u8
const fadt_pm1a_cnt_blk = 64; // u32 (I/O port)
const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
const fadt_pm1_cnt_len = 89; // u8 (bytes)
const fadt_sci_int = 46; // u16 (the SCI's GSI)
const fadt_flags = 112; // u32
const fadt_reset_register = 116; // GAS (12 bytes)
const fadt_reset_value = 128; // u8
const fadt_x_dsdt = 140; // u64
const fadt_x_pm1a_cnt_blk = 172; // GAS
const fadt_x_pm1b_cnt_blk = 184; // GAS
const fadt_x_pm_tmr_blk = 208; // GAS
const flag_reset_register_supported = 1 << 10;
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
/// FADT -> the power register map (into `power_information`) and the DSDT address,
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
const base: [*]align(1) const u8 = @ptrCast(header);
const len: usize = header.length;
const pi = &power_information;
pi.sci_interrupt = @truncate(fadt(u16, base, len, fadt_sci_int) orelse 0);
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
const cnt_width = fadt(u8, base, len, fadt_pm1_cnt_len) orelse 2;
pi.pm1a_cnt = readCntRegister(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width);
pi.pm1b_cnt = readCntRegister(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width);
const flags = fadt(u32, base, len, fadt_flags) orelse 0;
pi.reset_supported = flags & flag_reset_register_supported != 0;
pi.reset = readGas(base, len, fadt_reset_register) orelse .{};
pi.reset_value = fadt(u8, base, len, fadt_reset_value) orelse 0;
// The PM timer — a fixed-rate counter used as a calibration reference when no
// HPET is present. Prefer the 64-bit-capable X_ GAS, fall back to the port.
platform_information.pm_timer = readCntRegister(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4);
platform_information.pm_timer_32bit = flags & flag_tmr_value_ext != 0;
var dsdt: u64 = fadt(u32, base, len, fadt_dsdt) orelse 0;
if (fadt(u64, base, len, fadt_x_dsdt)) |x| {
if (x != 0) dsdt = x;
}
dsdt_physical = dsdt;
addAmlBlock(dsdt);
}
// SPCR field offsets (bytes from the table start).
const spcr_interface_type = 36; // u8
const spcr_base_address = 40; // GAS (12 bytes)
/// SPCR -> the console UART's address + interface type, so serial can target the
/// firmware's actual debug port instead of assuming legacy COM1.
fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
const base: [*]align(1) const u8 = @ptrCast(header);
const len: usize = header.length;
const gas = readGas(base, len, spcr_base_address) orelse return;
if (gas.address == 0) return;
platform_information.spcr_uart = gas;
platform_information.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0;
}
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
// register base sits at offset 8 within it.
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
const dmar_type_drhd: u16 = 0;
const drhd_register_base_offset = 8;
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
/// register block, and record its version and capabilities. This is *detection only*:
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
/// be gated by a per-device translation domain), but no domains are programmed yet —
/// enforcement is built with the first DMA driver, which is what there is to protect and
/// test against. See docs/driver-model.md (M16), the honest caveat.
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
const base: [*]align(1) const u8 = @ptrCast(header);
const total: usize = header.length;
var off: usize = dmar_structures_offset;
while (off + 4 <= total) {
const kind = fadt(u16, base, total, off) orelse break;
const length = fadt(u16, base, total, off + 2) orelse break;
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
if (kind == dmar_type_drhd) {
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
if (register_base != 0) {
const regs = hal.mapMmio(register_base, abi.page_size, true);
platform_information.iommu_present = true;
platform_information.iommu_base = register_base;
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
return; // first unit is enough for detection; multi-unit is future
}
}
off += length;
}
}
// --- helpers ----------------------------------------------------------------
/// Sum `len` bytes; an ACPI table/pointer is valid when the low 8 bits are zero.
fn checksumOk(bytes: [*]const u8, len: usize) bool {
var sum: u8 = 0;
for (0..len) |i| sum +%= bytes[i];
return sum == 0;
}
/// Read a FADT field of type `T` at `off`, or null if the table is too short to
/// contain it (a legal state for older FADT revisions).
fn fadt(comptime T: type, base: [*]align(1) const u8, len: usize, off: usize) ?T {
if (off + @sizeOf(T) > len) return null;
return rd(T, base, off);
}
/// Decode a Generic Address Structure at `off` into a `RegisterAccess`. GAS layout:
/// address_space(u8), bit_width(u8), bit_offset(u8), access_size(u8), address(u64).
fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegisterAccess {
if (off + 12 > len) return null;
const address_space = rd(u8, base, off);
const bit_width = rd(u8, base, off + 1);
const address = rd(u64, base, off + 4);
return .{
.mmio = address_space == 0, // 0 = system memory, 1 = system I/O
.address = address,
.width = bit_width / 8,
};
}
/// A PM1 control register: prefer the 64-bit-capable X_ GAS form; fall back to the
/// legacy 32-bit I/O-port field. Width comes from PM1_CNT_LEN either way.
fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegisterAccess {
if (readGas(base, len, xoff)) |g| {
if (g.address != 0) return .{ .mmio = g.mmio, .address = g.address, .width = width };
}
const port = fadt(u32, base, len, legacy_off) orelse 0;
return .{ .mmio = false, .address = port, .width = width };
}
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
/// x86 is little-endian and native, so an unaligned load suffices.
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
const p: *align(1) const T = @ptrCast(bytes + off);
return p.*;
}