Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
198767bb30 | ||
|
|
6a687fbc2b | ||
|
|
4e7cbc9792 | ||
|
|
e94adcfc02 | ||
|
|
f477ef7d9f | ||
|
|
9e649178bf | ||
|
|
e376c9e908 | ||
|
|
48b9ed4001 | ||
|
|
081ba1d74e | ||
|
|
203528c8a7 | ||
|
|
bf0c3fd3e0 | ||
|
|
3af0110483 | ||
|
|
757c6f14c3 |
@@ -63,7 +63,7 @@ zig build release-x86-64
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/os-development-guide/release-iso.md). `zig build check-iso-image`
|
||||
[docs/release-iso.md](docs/os-development/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
@@ -191,6 +191,7 @@ fn addKernel(
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.strip = optimize != .Debug, // DWARF info doubles the flashable image; keep it only for debug builds
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||
.{ .name = "abi", .module = modules.abi },
|
||||
@@ -283,6 +284,19 @@ pub fn build(b: *std.Build) void {
|
||||
const pci_class_module = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("library/device/pci/pci-class.zig"),
|
||||
});
|
||||
// Shared CSV helpers (comment stripping, field iteration) for the /etc/*.csv
|
||||
// config files — the device registry and the init service list both parse them.
|
||||
const csv_module = b.addModule("csv", .{
|
||||
.root_source_file = b.path("library/csv/csv.zig"),
|
||||
});
|
||||
// The device registry: parse /etc/devices.csv into match rules and bind a
|
||||
// reported device to a driver — the data-driven, authoritative replacement for
|
||||
// the manager's hand-written switch tables. Pure logic (no hardware, no
|
||||
// syscalls), so it unit-tests on the host; the manager imports it.
|
||||
const device_registry_module = b.addModule("device-registry", .{
|
||||
.root_source_file = b.path("library/device/registry/device-registry.zig"),
|
||||
.imports = &.{.{ .name = "csv", .module = csv_module }},
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||
@@ -517,15 +531,18 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header fields, BAR
|
||||
// decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI
|
||||
// decode + map, capability walks (legacy + extended), MSI/MSI-X programming, power
|
||||
// state, and function-level reset (library/device/pci/pci.zig). The generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline. Imports the driver (device
|
||||
// access) client + mmio + the pci-class data module (config-space layout constants).
|
||||
// access) client + mmio + the pci-class data module (config-space layout constants) +
|
||||
// time (the D0 and FLR settle delays).
|
||||
const pci_module = b.addModule("pci", .{
|
||||
.root_source_file = b.path("library/device/pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver_module },
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
.{ .name = "pci-class", .module = pci_class_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
},
|
||||
});
|
||||
|
||||
@@ -638,6 +655,8 @@ pub fn build(b: *std.Build) void {
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, &default_imports, "init", "system/services/init/init.zig");
|
||||
programModule(init_exe).addImport("power-protocol", power_protocol_module);
|
||||
// init parses its boot service list from /etc/init.csv with the shared csv helpers.
|
||||
programModule(init_exe).addImport("csv", csv_module);
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat is a
|
||||
// serial/test-build diagnostic (the QEMU harness's init tests assert on it, and
|
||||
// -Dserial images emit it), so a flashable image runs a purely event-driven PID 1
|
||||
@@ -645,7 +664,8 @@ pub fn build(b: *std.Build) void {
|
||||
// the heartbeat stays present under test.
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
init_options.addOption(bool, "diagnose", diagnose);
|
||||
// Which services init starts is no longer a comptime option: it reads /etc/init.csv,
|
||||
// and -Ddiagnose selects which init.csv is bundled (see the `bundled` list below).
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
|
||||
// --- the rest of the boot tree: /system services and drivers, /test fixtures ---
|
||||
@@ -660,6 +680,8 @@ pub fn build(b: *std.Build) void {
|
||||
programModule(ps2_mouse_exe).addImport("input-protocol", input_protocol_module);
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
programModule(usb_xhci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// MSI setup: the claimed-function view (enable bits + MSI capability programming).
|
||||
programModule(usb_xhci_bus_exe).addImport("pci", pci_module);
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
@@ -710,6 +732,16 @@ pub fn build(b: *std.Build) void {
|
||||
programModule(crash_test_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
const device_list_exe = addUserBinary(b, kernel_target, &default_imports, "device-list", "test/system/services/device-list/device-list.zig");
|
||||
programModule(device_list_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// A test fixture: claims the pci-caps case's extra unclaimed NIC and exercises the
|
||||
// driver-side PCI library surface (capabilities, MSI, MSI-X, power, FLR) against it.
|
||||
const pci_cap_test_exe = addUserBinary(b, kernel_target, &default_imports, "pci-cap-test", "test/system/services/pci-cap-test/pci-cap-test.zig");
|
||||
programModule(pci_cap_test_exe).addImport("pci", pci_module);
|
||||
programModule(pci_cap_test_exe).addImport("pci-class", pci_class_module);
|
||||
// The IOMMU-enforcement negative test: claims an unclaimed e1000e and fires a rogue
|
||||
// DMA that VT-d must fault. Same PCI building blocks as pci-cap-test.
|
||||
const iommu_fault_test_exe = addUserBinary(b, kernel_target, &default_imports, "iommu-fault-test", "test/system/services/iommu-fault-test/iommu-fault-test.zig");
|
||||
programModule(iommu_fault_test_exe).addImport("pci", pci_module);
|
||||
programModule(iommu_fault_test_exe).addImport("pci-class", pci_class_module);
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
@@ -728,12 +760,10 @@ pub fn build(b: *std.Build) void {
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("power-protocol", power_protocol_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, &default_imports, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
programModule(device_manager_exe).addImport("pci-class", pci_class_module);
|
||||
programModule(device_manager_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
programModule(device_manager_exe).addImport("usb-ids", usb_ids_module);
|
||||
// Driver matching is data-driven: the manager parses /etc/devices.csv into this
|
||||
// module's rules and binds each reported device by most-specific match.
|
||||
programModule(device_manager_exe).addImport("device-registry", device_registry_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, &default_imports, "input", "system/services/input/input.zig");
|
||||
@@ -754,7 +784,11 @@ pub fn build(b: *std.Build) void {
|
||||
// structure is the single source of truth. Entry names (and hence argv[0] and
|
||||
// task names) are these paths with a leading slash. Test fixtures mirror their
|
||||
// repo home: test/system/services/<name> in the source tree IS the boot path.
|
||||
const bundled = [_]BundledBinary{
|
||||
// init's boot service list is data (/etc/init.csv). -Ddiagnose selects the
|
||||
// variant that omits the display stack (so the kernel's boot transcript stays
|
||||
// on screen); both are bundled at the same /etc/init.csv path.
|
||||
const init_csv_source = if (diagnose) "etc/init-diagnose.csv" else "etc/init.csv";
|
||||
const production_bundled = [_]BundledBinary{
|
||||
.{ .path = "system/services/init", .binary = init_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display", .binary = display_exe.getEmittedBin() },
|
||||
@@ -763,6 +797,13 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .path = "system/services/input", .binary = input_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/discovery", .binary = discovery_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/logger", .binary = logger_exe.getEmittedBin() },
|
||||
// A data file, not a binary: the device registry the manager reads at boot.
|
||||
// Packing it under /etc makes the kernel auto-mount /etc as a read-only
|
||||
// initrd tree (system/kernel/vfs.zig setInitialRamdisk), so the manager can
|
||||
// fs.open("/etc/devices.csv") with no filesystem service running.
|
||||
.{ .path = "etc/devices.csv", .binary = b.path("etc/devices.csv") },
|
||||
// init's service list, likewise read from the kernel-served initrd /etc.
|
||||
.{ .path = "etc/init.csv", .binary = b.path(init_csv_source) },
|
||||
.{ .path = "system/drivers/ps2-bus", .binary = ps2_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-keyboard", .binary = ps2_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-mouse", .binary = ps2_mouse_exe.getEmittedBin() },
|
||||
@@ -772,18 +813,34 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .path = "system/drivers/usb-storage", .binary = usb_storage_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/virtio-gpu", .binary = virtio_gpu_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/pci-bus", .binary = pci_bus_exe.getEmittedBin() },
|
||||
};
|
||||
// The userspace test fixtures under /test. A plain `zig build` produces a clean
|
||||
// image WITHOUT them; they are bundled only for a test build — which the QEMU
|
||||
// harness signals by passing -Dtest-case=<name> for every scenario, exactly when
|
||||
// these fixtures must be on the boot volume. Merely building this array never
|
||||
// forces a compile: the fixture exes build only if `bundled` (below) includes them.
|
||||
const test_bundled = [_]BundledBinary{
|
||||
.{ .path = "test/system/services/vfs-test", .binary = vfstest_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/fat-test", .binary = fat_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/shared-memory-server", .binary = shared_memory_server_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/pci-cap-test", .binary = pci_cap_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/iommu-fault-test", .binary = iommu_fault_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/process-test", .binary = process_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/thread-test", .binary = thread_test_exe.getEmittedBin() },
|
||||
};
|
||||
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
||||
// production set only. Test fixtures join in only under -Dtest-case; the
|
||||
// diagnose display-omission is already handled by init_csv_source above.
|
||||
var bundled_list: std.ArrayListUnmanaged(BundledBinary) = .empty;
|
||||
bundled_list.appendSlice(b.allocator, &production_bundled) catch @panic("OOM");
|
||||
if (test_case != null) bundled_list.appendSlice(b.allocator, &test_bundled) catch @panic("OOM");
|
||||
const bundled = bundled_list.items;
|
||||
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||
@@ -866,7 +923,7 @@ pub fn build(b: *std.Build) void {
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, &bundled);
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
@@ -875,7 +932,7 @@ pub fn build(b: *std.Build) void {
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, &bundled);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
@@ -1077,6 +1134,7 @@ pub fn build(b: *std.Build) void {
|
||||
"library/device/acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"library/device/usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"library/device/usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/csv/csv.zig", // shared /etc/*.csv comment-strip + field-split helpers
|
||||
"library/device/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
@@ -1101,6 +1159,19 @@ pub fn build(b: *std.Build) void {
|
||||
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||
}
|
||||
|
||||
// The device registry imports the shared `csv` module, so its tests need that
|
||||
// import wired and don't fit the plain loop above. These prove the /etc/devices.csv
|
||||
// parse + most-specific driver match (incl. virtio 1AF4:1050 beating a class rule).
|
||||
const device_registry_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/device/registry/device-registry.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{.{ .name = "csv", .module = csv_module }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(device_registry_tests).step);
|
||||
|
||||
// The xkeyboard-config keymap tests need its generated `layouts` import wired, so they
|
||||
// don't fit the plain loop above. Its keycode->character assertions are the end-to-end
|
||||
// proof that the xkb-data -> generator -> Zig-lookup pipeline is correct.
|
||||
|
||||
+66
-64
@@ -3,102 +3,104 @@
|
||||
Notes on how danos boots and draws, written to explain the *why* behind the code
|
||||
rather than restate it. Roughly in the order things happen at runtime:
|
||||
|
||||
1. **[efi.md](os-development-guide/efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
1. **[efi.md](os-development/efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
runs the bootloader, what the loader gathers before `ExitBootServices`, how it
|
||||
loads the kernel ELF, and the ABI contract for the jump into the kernel. Start
|
||||
here.
|
||||
2. **[system-image.md](os-development-guide/system-image.md) — system.img, the boot capsule.** The
|
||||
2. **[system-image.md](os-development/system-image.md) — system.img, the boot capsule.** The
|
||||
bundled user binaries packed into one file in the initial-ramdisk wire
|
||||
format, because one open + one sequential read is the only file I/O shape
|
||||
firmware is fast at. The trivial container format, the three artifacts one
|
||||
build list derives (tree, manifest, capsule), the loader's three-strategy
|
||||
fallback chain, and the capsule's kernel-side life as both the spawn table
|
||||
and the read-only `/system` mount.
|
||||
3. **[gop.md](os-development-guide/gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
3. **[gop.md](os-development/gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
modes (unlike fixed VGA modes), how we detect the monitor's native resolution
|
||||
from EDID and switch to it, and the pixel formats we accept or reject.
|
||||
4. **[framebuffer.md](os-development-guide/framebuffer.md) — the framebuffer.** What the linear
|
||||
4. **[framebuffer.md](os-development/framebuffer.md) — the framebuffer.** What the linear
|
||||
framebuffer the loader hands over actually is, and what **pitch** (stride)
|
||||
means versus width — the detail you have to get right to avoid a skewed image.
|
||||
5. **[memory-map.md](os-development-guide/memory-map.md) — the memory map.** How the loader learns what
|
||||
5. **[memory-map.md](os-development/memory-map.md) — the memory map.** How the loader learns what
|
||||
physical RAM exists and hands it to the kernel in danos's own neutral format,
|
||||
rather than leaking UEFI's memory descriptors across the boundary.
|
||||
6. **[frame-allocator.md](os-development-guide/frame-allocator.md) — the physical frame allocator.** The
|
||||
6. **[frame-allocator.md](os-development/frame-allocator.md) — the physical frame allocator.** The
|
||||
bitmap allocator that hands out and reclaims 4 KiB physical frames from that
|
||||
map — the primitive page tables and the heap are built on.
|
||||
7. **[interrupts.md](os-development-guide/interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
7. **[interrupts.md](os-development/interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
TSS, the exception stubs, and the handler that reports a CPU fault in red instead
|
||||
of letting it triple-fault into a silent reset.
|
||||
8. **[paging.md](os-development-guide/paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
8. **[paging.md](os-development/paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
page tables, identity-mapping the low 4 GiB, and switching CR3 off the firmware's
|
||||
tables onto ours.
|
||||
9. **[device-interrupts.md](device-driver-development-guide/device-interrupts.md) — device interrupts.** The Local
|
||||
9. **[device-interrupts.md](device-driver-development/device-interrupts.md) — device interrupts.** The Local
|
||||
APIC and its timer — the kernel's first interrupt that is *handled and returned
|
||||
from*, giving it a heartbeat.
|
||||
10. **[heap.md](os-development-guide/heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
10. **[heap.md](os-development/heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
the VMM, exposed as a `std.mem.Allocator` so std containers work — dynamic
|
||||
allocation for the kernel.
|
||||
11. **[scheduling.md](os-development-guide/scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
11. **[scheduling.md](os-development/scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||
blocking (sleep, wait queues) — the leap to a running system.
|
||||
12. **[ipc.md](device-driver-development-guide/ipc.md) — inter-process communication.** Bounded blocking
|
||||
12. **[ipc.md](device-driver-development/ipc.md) — inter-process communication.** Bounded blocking
|
||||
message-passing channels, then synchronous call/reply between *processes* over
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
13. **[syscall.md](os-development-guide/syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
13. **[syscall.md](os-development/syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](os-development-guide/vdso.md) designs
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](os-development/vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
14. **[vfs-protocol.md](file-system-development/vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
15. **[drivers.md](device-driver-development-guide/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
15. **[drivers.md](device-driver-development/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
16. **[driver-model.md](device-driver-development-guide/driver-model.md) — buses, classes and host controllers.** How
|
||||
unmask. The condensed version: the
|
||||
[new-driver checklist](device-driver-development/new-driver-checklist.md) —
|
||||
the minimum steps from boot-log line to mapped registers.
|
||||
16. **[driver-model.md](device-driver-development/driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
17. **[usb-hub.md](device-driver-development-guide/usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
17. **[usb-hub.md](device-driver-development/usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
*inside* the `usb-xhci-bus` driver rather than a separate hub class driver — a
|
||||
device behind a hub is reached by the **controller**, programmed with a route
|
||||
string in its slot context — plus the compound-hub reality (a USB 3.0 hub is
|
||||
physically two hubs) and detection via the hub's status-change interrupt endpoint.
|
||||
18. **[process-management.md](os-development-guide/process-management.md) — process management.** The
|
||||
18. **[process-management.md](os-development/process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
19. **[process-lifecycle.md](os-development-guide/process-lifecycle.md) — the process lifecycle.** Built
|
||||
19. **[process-lifecycle.md](os-development/process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`process` module interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
20. **[device-manager.md](device-driver-development-guide/device-manager.md) — the device manager.** Built (M18,
|
||||
20. **[device-manager.md](device-driver-development/device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](os-development-guide/resilience.md)'s restart goal into increments.
|
||||
21. **[input.md](device-driver-development-guide/input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
[resilience.md](os-development/resilience.md)'s restart goal into increments.
|
||||
21. **[input.md](device-driver-development/input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
22. **[display.md](device-driver-development-guide/display.md) — the display service.** The display half of the GUI
|
||||
22. **[display.md](device-driver-development/display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](device-driver-development-guide/display-plan.md). **v2** (complete) makes
|
||||
that tear-free doesn't. Plan: [display-plan.md](device-driver-development/display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](device-driver-development-guide/display-v2.md), plan [display-v2-plan.md](device-driver-development-guide/display-v2-plan.md). Looking
|
||||
[display-v2.md](device-driver-development/display-v2.md), plan [display-v2-plan.md](device-driver-development/display-v2-plan.md). Looking
|
||||
further out, three research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](device-driver-development-guide/nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere), [amd-gpus.md](device-driver-development-guide/amd-gpus.md) (RX 6600 / RDNA2), and [intel-igpu.md](device-driver-development-guide/intel-igpu.md)
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](device-driver-development/nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere), [amd-gpus.md](device-driver-development/amd-gpus.md) (RX 6600 / RDNA2), and [intel-igpu.md](device-driver-development/intel-igpu.md)
|
||||
(Intel iGPU).
|
||||
23. **[halting.md](os-development-guide/halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
23. **[halting.md](os-development/halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -108,7 +110,7 @@ Start with the north star:
|
||||
**resilience** (restartable components). Win condition: runs on the author's PC and
|
||||
both Raspberry Pis, ideally with a GUI. Real-time is an option to explore, not a
|
||||
requirement. The *why* that shapes everything below.
|
||||
- **[resilience.md](os-development-guide/resilience.md) — resilience.** A design note (not built yet) on
|
||||
- **[resilience.md](os-development/resilience.md) — resilience.** A design note (not built yet) on
|
||||
fault isolation + live restart — the reincarnation-server + capability model that
|
||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||
@@ -117,14 +119,14 @@ Start with the north star:
|
||||
port to **one seam** (`std.os.danos`), so we build an `os` seam module (→ that seam) plus
|
||||
the thin `file-system` module, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](os-development-guide/threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
- **[threading.md](os-development/threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
the `thread` module's `Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](os-development-guide/syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](os-development-guide/resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](os-development-guide/threading-plan.md).
|
||||
- **[vdso.md](os-development-guide/vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
native type and not literal `std.Thread` (the [private ABI](os-development/syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](os-development/resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](os-development/threading-plan.md).
|
||||
- **[vdso.md](os-development/vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
@@ -137,69 +139,69 @@ Cutting across all of these:
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](os-development-guide/release-iso.md) — the release ISO.** The flashable boot
|
||||
- **[release-iso.md](os-development/release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[architecture.md](os-development-guide/architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
- **[architecture.md](os-development/architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
- **[arm.md](os-development-guide/arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
- **[arm.md](os-development/arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
aiming at: `arm` (32-bit, Pi Zero W) vs `aarch64` (64-bit, Pi 3-5), UEFI vs
|
||||
device-tree boot, and what each layer needs.
|
||||
- **[discovery.md](os-development-guide/discovery.md) — device discovery.** A design note on learning what
|
||||
- **[discovery.md](os-development/discovery.md) — device discovery.** A design note on learning what
|
||||
hardware exists via ACPI (x86) or device tree (ARM) behind one neutral device model —
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](os-development-guide/acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
- **[acpi.md](os-development/acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInformation`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](os-development-guide/power.md) — the power service.** System power as a domain-named
|
||||
- **[power.md](os-development/power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](os-development-guide/process-lifecycle.md) stop sequence with an ACPI
|
||||
shutdown composing the [lifecycle](os-development/process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](os-development-guide/timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
- **[timers.md](os-development/timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-driver-development-guide/device-interrupts.md).
|
||||
- **[smp.md](os-development-guide/smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-driver-development/device-interrupts.md).
|
||||
- **[smp.md](os-development/smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||
- **[sysv.md](os-development-guide/sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
- **[sysv.md](os-development/sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||
QEMU and asserting on its serial output — reproducibly, and structured so the
|
||||
same tests run across architectures.
|
||||
- **[logging.md](os-development-guide/logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
- **[logging.md](os-development/logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
0xE9 debugcon, file later) kept separate from the framebuffer display, plus the
|
||||
robustness path: optional framebuffer, POST-code checkpoints, and a persistent
|
||||
panic breadcrumb so the kernel survives — and can be diagnosed — with no output.
|
||||
|
||||
## How the pieces relate
|
||||
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](os-development-guide/efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](os-development-guide/gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](os-development-guide/framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](os-development-guide/memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](os-development-guide/frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](os-development-guide/interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](os-development-guide/paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](os-development-guide/heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](os-development-guide/scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-driver-development-guide/device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](device-driver-development-guide/ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](os-development-guide/architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](os-development-guide/halting.md)).
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](os-development/efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](os-development/gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](os-development/framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](os-development/memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](os-development/frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](os-development/interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](os-development/paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](os-development/heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](os-development/scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-driver-development/device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](device-driver-development/ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](os-development/architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](os-development/halting.md)).
|
||||
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](os-development-guide/discovery.md),
|
||||
[acpi.md](os-development-guide/acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](os-development-guide/syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](device-driver-development-guide/ipc.md)), and a **[driver](device-driver-development-guide/drivers.md)** claims
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](os-development/discovery.md),
|
||||
[acpi.md](os-development/acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](os-development/syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](device-driver-development/ipc.md)), and a **[driver](device-driver-development/drivers.md)** claims
|
||||
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
|
||||
|
||||
+8
-8
@@ -1,6 +1,6 @@
|
||||
# Device interrupts
|
||||
|
||||
CPU exceptions ([interrupts.md](../os-development-guide/interrupts.md)) are the kernel reacting to its own
|
||||
CPU exceptions ([interrupts.md](../os-development/interrupts.md)) are the kernel reacting to its own
|
||||
mistakes. **Device interrupts** are the opposite: hardware asking for attention —
|
||||
a timer firing, a key pressed, a packet arriving. They share the IDT, but differ
|
||||
in one fundamental way: an exception here is terminal (we report and halt), while a
|
||||
@@ -10,7 +10,7 @@ back — the same mechanism a scheduler will later use to preempt tasks.
|
||||
|
||||
The first device we bring up is the **timer**, because it's the simplest: it lives
|
||||
entirely on the CPU's local interrupt controller, needing no external routing.
|
||||
It's all x86_64-specific, behind the [architecture](../os-development-guide/architecture.md) boundary.
|
||||
It's all x86_64-specific, behind the [architecture](../os-development/architecture.md) boundary.
|
||||
|
||||
## The APIC, not the PIC
|
||||
|
||||
@@ -53,7 +53,7 @@ a missing PIT would hang the boot):
|
||||
|
||||
1. **CPUID leaf 0x15** — the CPU's TSC frequency directly, needing no external timer
|
||||
at all (the LAPIC is then measured against the TSC).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](../os-development-guide/discovery.md) / [acpi](../os-development-guide/acpi.md)).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](../os-development/discovery.md) / [acpi](../os-development/acpi.md)).
|
||||
3. The **ACPI PM timer** (a fixed 3.579545 MHz counter from the FADT).
|
||||
4. The **PIT** (legacy 8254, 1.193182 MHz) — last resort, and bounded so it can't hang.
|
||||
|
||||
@@ -100,7 +100,7 @@ values (a second socket, some firmware), so a thread migrating from a core readi
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](../os-development-guide/smp.md)).
|
||||
come up one at a time ([smp.md](../os-development/smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
@@ -160,7 +160,7 @@ A device handler is a plain `fn () void` — a timer or keyboard handler doesn't
|
||||
the interrupted registers. (The stubs originally didn't save the SSE/vector
|
||||
registers, so a handler couldn't use them; `isr_common` now does an
|
||||
`fxsave`/`fxrstor` of the full SSE/x87 state around dispatch — see
|
||||
[interrupts.md](../os-development-guide/interrupts.md).)
|
||||
[interrupts.md](../os-development/interrupts.md).)
|
||||
|
||||
## Turning them on
|
||||
|
||||
@@ -168,7 +168,7 @@ Exceptions can't be masked, which is why they worked all along. Maskable device
|
||||
interrupts don't fire until the CPU's interrupt flag is set — so the final step is
|
||||
`sti` (`arch.enableInterrupts()`), after the APIC and timer are configured. From
|
||||
that instant the kernel has a heartbeat, and its idle `hlt` loop
|
||||
([halting.md](../os-development-guide/halting.md)) wakes on every tick and dozes off again.
|
||||
([halting.md](../os-development/halting.md)) wakes on every tick and dozes off again.
|
||||
|
||||
## Verifying it
|
||||
|
||||
@@ -188,13 +188,13 @@ spinning in unrelated code — is the whole mechanism working end to end.
|
||||
## Since (done elsewhere)
|
||||
|
||||
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||
reason a *returning* interrupt matters. See [scheduling.md](../os-development-guide/scheduling.md).
|
||||
reason a *returning* interrupt matters. See [scheduling.md](../os-development/scheduling.md).
|
||||
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||
[drivers.md](drivers.md).
|
||||
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||
user drivers — see [paging.md](../os-development-guide/paging.md).
|
||||
user drivers — see [paging.md](../os-development/paging.md).
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
+24
-15
@@ -10,18 +10,18 @@ mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](../os-development-guide/process-management.md):
|
||||
The primitives underneath are real ([process-management.md](../os-development/process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||
policy process that turns [resilience.md](../os-development-guide/resilience.md)'s restart goal into practice
|
||||
policy process that turns [resilience.md](../os-development/resilience.md)'s restart goal into practice
|
||||
for drivers.
|
||||
|
||||
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||
document: that is the universal lifecycle every danos process speaks —
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md), signals over IPC and the stable
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md), signals over IPC and the stable
|
||||
`process` interface. The device manager is that design's first serious
|
||||
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||
driver is stopped, health-checked, and buried exactly like any other process.
|
||||
@@ -35,7 +35,7 @@ The device tree is two things fused: *information* (what exists, how it nests) a
|
||||
claims, resource containment on `device_register`, the
|
||||
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||
dies** (settled; it is increment 1 of
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md)). The three invariants in
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)). The three invariants in
|
||||
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||
buggy one would un-earn everything the microkernel bought.
|
||||
@@ -51,7 +51,7 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands. (It landed: [discovery.md](../os-development-guide/discovery.md), M19–M20.)
|
||||
depends on when it lands. (It landed: [discovery.md](../os-development/discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
@@ -82,7 +82,7 @@ one world.
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
@@ -91,13 +91,16 @@ mechanism), replacing first-come-first-served `device_claim` with policy. Identi
|
||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
(Since the registry landed, `child_added` also carries a `bus` discriminator and
|
||||
the numeric `vendor`/`device`/`subsystem` ids the finer match levels need —
|
||||
see [/etc/devices.csv](devices-csv.md).)
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||
On a death notification:
|
||||
|
||||
1. **Read the reason** ([process-lifecycle.md](../os-development-guide/process-lifecycle.md) increment 2).
|
||||
1. **Read the reason** ([process-lifecycle.md](../os-development/process-lifecycle.md) increment 2).
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
@@ -136,7 +139,7 @@ way.
|
||||
## Increments
|
||||
|
||||
Increments 1–4 are the lifecycle prerequisites and live in
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `process`). On top of those:
|
||||
|
||||
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||
@@ -147,7 +150,7 @@ published exit events, signals + `process`). On top of those:
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](../os-development-guide/discovery.md); of the enumerable
|
||||
the acpi service (M20), see [discovery.md](../os-development/discovery.md); of the enumerable
|
||||
devices, the kernel seeds only the host bridge and the acpi-tables node (the
|
||||
non-enumerable platform nodes — processors, interrupt controllers, the HPET,
|
||||
the loader's framebuffer — stay kernel-seeded too). Matching moved with it:
|
||||
@@ -169,9 +172,15 @@ published exit events, signals + `process`). On top of those:
|
||||
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||
the world (above). Checkpointing driver state with the manager is deferred until
|
||||
something demonstrates the need.
|
||||
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` were
|
||||
honest at two bus types; the third was expected to trigger the manifest (a driver
|
||||
declares what it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||
(Since then: the third bus — USB — arrived and is matched in code too. Today's
|
||||
matchers are `pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity`;
|
||||
the manifest waits until code matching actually hurts.)
|
||||
- **Matching is a registry, not code (resolved 2026-07-26).** `driverFor`/
|
||||
`pciDriverFor` were honest at two bus types; the third (USB) was matched in code
|
||||
too, and then the switch tables started to hurt — they keyed PCI matches on the
|
||||
class triple alone, so a virtio-gpu could only be matched as a generic display
|
||||
function and the driver had to re-confirm its `1AF4:1050` identity from config
|
||||
space after being spawned. The manifest the earlier note anticipated landed as a
|
||||
human-readable registry: **[/etc/devices.csv](devices-csv.md)**, parsed by the
|
||||
pure `device-registry` module and read by the manager at boot. A row binds a
|
||||
driver to a device by any of base / subclass / prog-IF / vendor / device /
|
||||
subsystem / `_HID`, most-specific match winning; it is authoritative (no
|
||||
compiled-in fallback — an unmatched device is logged, never guessed).
|
||||
`pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity` are gone.
|
||||
@@ -0,0 +1,102 @@
|
||||
# /etc/devices.csv — the device registry
|
||||
|
||||
**Status: built (2026-07-26).** The device manager reads `/etc/devices.csv` at
|
||||
boot and binds every device a bus driver reports to the driver the registry
|
||||
names. It replaces the three hand-written `switch` tables that used to live in
|
||||
the manager (`pciDriverForIdentity`, `hidDriverFor`, `usbDriverForIdentity`) —
|
||||
the "manifest" [device-manager.md](device-manager.md) anticipated once code
|
||||
matching started to hurt. The parser and matcher are the pure, unit-tested
|
||||
`device-registry` module (`library/device/registry/device-registry.zig`).
|
||||
|
||||
## Why a registry
|
||||
|
||||
The switch tables keyed PCI matches on the 24-bit class/subclass/prog-IF triple
|
||||
alone. That is too coarse: a virtio-gpu is just "display / other" by class, so it
|
||||
could only be *class-matched* and the driver had to re-confirm its real
|
||||
`1AF4:1050` identity from config space **after** the manager had already spawned
|
||||
it. The registry lets a rule bind on the full identity — down to vendor, device,
|
||||
and subsystem — so the manager makes the precise decision itself, and the driver
|
||||
comes up already knowing it is the right one.
|
||||
|
||||
It is also **data, not code**: teaching the system new hardware is a line in a
|
||||
file, not an edit-and-recompile of the manager. And it is **greppable** — one
|
||||
place to read "what binds what," the same idea as Linux's `modules.alias`.
|
||||
|
||||
## The file
|
||||
|
||||
One rule per line, nine comma-separated fields; `#` starts a comment (whole-line
|
||||
or trailing); blank lines are ignored. Whitespace around a field is trimmed, so
|
||||
columns may be padded for readability.
|
||||
|
||||
```
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
```
|
||||
|
||||
| Field | Meaning | Notes |
|
||||
|---|---|---|
|
||||
| `bus` | `pci` \| `usb` \| `acpi` | which bus reported the device; picks the namespace for the id columns |
|
||||
| `base` | PCI base class / USB class | hex |
|
||||
| `class` | PCI subclass / USB subclass | hex |
|
||||
| `prog_if` | PCI prog-IF / USB protocol | hex |
|
||||
| `vendor` | PCI vendor / USB idVendor | hex |
|
||||
| `device` | PCI device / USB idProduct | hex |
|
||||
| `subsystem` | PCI subsystem, `(ssvid<<16)\|ssid` | hex; blank for usb/acpi |
|
||||
| `hid` | ACPI `_HID` (e.g. `PNP0303`) | blank for pci/usb |
|
||||
| `driver` | full ramdisk path to spawn | e.g. `/system/drivers/virtio-gpu` |
|
||||
|
||||
`*` or an empty field is a **wildcard** — it matches anything and adds nothing to
|
||||
a rule's specificity.
|
||||
|
||||
## Levels of detection: most-specific-wins
|
||||
|
||||
Several rows may match one device. The manager picks the **most specific** — the
|
||||
one that pins the finest-grained fields. Specificity weights double from the
|
||||
coarsest level so each outweighs all coarser levels combined:
|
||||
|
||||
```
|
||||
base(1) < class(2) < prog_if(4) < vendor(8) < subsystem(16) < device(32) ≈ hid(32)
|
||||
```
|
||||
|
||||
So the generic `pci, 03, 00, 00, …/display` rule and the precise
|
||||
`pci, 03, 80, *, 1AF4, 1050, …/virtio-gpu` rule coexist: the virtio card
|
||||
(vendor 1AF4, device 1050) takes the specific rule; a plain VGA adapter still
|
||||
falls to the generic one. Two rules that match a device with the *same*
|
||||
specificity are a registry authoring error — the manager logs it loudly and binds
|
||||
the first, so the shadowed rule is visible rather than silently dropped.
|
||||
|
||||
## Authoritative — no code fallback
|
||||
|
||||
There is no compiled-in default table behind the registry. A device that no row
|
||||
matches goes **unbound** and is logged; the manager never guesses. A missing or
|
||||
empty `/etc/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
not a silent half-working system.
|
||||
|
||||
## How the manager reads it
|
||||
|
||||
`/etc/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/etc` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/etc` — so the manager
|
||||
reads the file with a plain `fs.open("/etc/devices.csv")` + `read`, with no
|
||||
filesystem service running and no boot-ordering dependency. It parses the bytes
|
||||
once in `initialise`, before any bus driver can report a device to match.
|
||||
|
||||
## Feeding the matcher: the widened report
|
||||
|
||||
Finer-grained matching needs identity the old ABI threw away. Two things carry it
|
||||
now: `child_added` (and `DeviceDescriptor`) grew `vendor` / `device` /
|
||||
`subsystem` fields, filled by the PCI bus driver from config space (offsets
|
||||
0x00 and 0x2C); and each bus driver states its `bus` in the report (a `BusKind`),
|
||||
so the manager reads a PCI class triple and a USB class triple — the same 24 bits
|
||||
in different namespaces — against the right `bus` column.
|
||||
|
||||
## Adding a driver
|
||||
|
||||
1. Build the driver binary and bundle it at `/system/drivers/<name>` (build.zig).
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
+1
-1
@@ -144,4 +144,4 @@ path in VMs**, where danos development happens. The framebuffer floor never goes
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](../os-development-guide/resilience.md) — the restart machinery the hot-attach leans on.
|
||||
- [resilience.md](../os-development/resilience.md) — the restart machinery the hot-attach leans on.
|
||||
+12
-12
@@ -1,6 +1,6 @@
|
||||
# The display service: a framebuffer compositor
|
||||
|
||||
The [framebuffer](../os-development-guide/framebuffer.md) the loader hands over is a flat block of pixel
|
||||
The [framebuffer](../os-development/framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../../system/kernel/console.zig) draws text
|
||||
into it directly. That console is a stop-gap. The **display service**
|
||||
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||
@@ -22,7 +22,7 @@ which one you're holding decides what you can do.
|
||||
linear framebuffer pointer and can set video modes — but only until
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../../boot/efi.zig)
|
||||
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||
exiting ([gop.md](../os-development-guide/gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
exiting ([gop.md](../os-development/gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
mode list, no EDID. What survives is the frozen snapshot in
|
||||
[`BootInformation.framebuffer`](../../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format, refresh_hz}`, and nothing more.
|
||||
@@ -66,7 +66,7 @@ rest of the system hasn't had to face:
|
||||
[`console.zig`](../../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) node, so
|
||||
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||
[syscall](../os-development-guide/syscall.md). A user-space display service needs a **new mechanism just to
|
||||
[syscall](../os-development/syscall.md). A user-space display service needs a **new mechanism just to
|
||||
touch the pixels**. (See "The handoff" below — this is built.)
|
||||
|
||||
2. **danos had no cross-process shared memory.** At v1 the memory syscalls were `mmap`
|
||||
@@ -119,7 +119,7 @@ second backend or a second monitor appears; until then it is complexity with no
|
||||
|
||||
The framebuffer crosses into user space through the machinery that already exists for
|
||||
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](../os-development-guide/resilience.md)
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](../os-development/resilience.md)
|
||||
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||
re-claims it).
|
||||
|
||||
@@ -160,8 +160,8 @@ Two buffers, with deliberately different memory types:
|
||||
|
||||
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||
[framebuffer](../os-development-guide/framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](../os-development-guide/gop.md).
|
||||
[framebuffer](../os-development/framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](../os-development/gop.md).
|
||||
|
||||
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||
|
||||
@@ -234,7 +234,7 @@ client module, as with every other danos service.
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](../os-development-guide/threading.md): the service is built
|
||||
ownership is the display's first use of [threads](../os-development/threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
@@ -242,7 +242,7 @@ thread** beside the compositor loop.
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](../os-development-guide/halting.md)).
|
||||
free to halt ([halting.md](../os-development/halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
@@ -254,13 +254,13 @@ thread** beside the compositor loop.
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](../os-development-guide/threading.md)). IPC **handles do
|
||||
Two threading facts shape this (both in [threading.md](../os-development/threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](../os-development-guide/resilience.md)).
|
||||
([resilience.md](../os-development/resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
@@ -317,8 +317,8 @@ packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
## See also
|
||||
|
||||
- [framebuffer.md](../os-development-guide/framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](../os-development-guide/gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [framebuffer.md](../os-development/framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](../os-development/gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||
+4
-4
@@ -34,7 +34,7 @@ plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDescriptor`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](../os-development-guide/discovery.md)); `device_register` grows it.
|
||||
seeds it ([discovery.md](../os-development/discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
@@ -199,7 +199,7 @@ class driver, the device manager, or the kernel may share them freely.
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||
([sysv.md](../os-development-guide/sysv.md)). This is what
|
||||
([sysv.md](../os-development/sysv.md)). This is what
|
||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||
@@ -398,6 +398,6 @@ from hand-rolling `*volatile` and getting ARM wrong.
|
||||
## See also
|
||||
|
||||
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||
- [discovery.md](../os-development-guide/discovery.md) / [acpi.md](../os-development-guide/acpi.md) — where the device table comes from.
|
||||
- [discovery.md](../os-development/discovery.md) / [acpi.md](../os-development/acpi.md) — where the device table comes from.
|
||||
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||
- [resilience.md](../os-development-guide/resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
- [resilience.md](../os-development/resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
+7
-5
@@ -4,13 +4,15 @@ In a monolithic kernel a driver is a function call away from everything: it runs
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](../os-development-guide/resilience.md)).
|
||||
can be restarted ([resilience](../os-development/resilience.md)). This document is
|
||||
the reasoning; the condensed do-this-then-that version is the
|
||||
[new-driver checklist](new-driver-checklist.md).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||
([discovery](../os-development-guide/discovery.md), [acpi](../os-development-guide/acpi.md)).
|
||||
([discovery](../os-development/discovery.md), [acpi](../os-development/acpi.md)).
|
||||
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||
device's physical MMIO window into your address space, and from then on it's plain
|
||||
memory. No syscall per register access.
|
||||
@@ -51,7 +53,7 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||
its probe path, entirely from ring 3:
|
||||
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||
ACPI/PCI ([discovery](../os-development-guide/discovery.md)).
|
||||
ACPI/PCI ([discovery](../os-development/discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver. The match policy is code, a few
|
||||
small per-bus tables: from the boot snapshot only the PCI host bridge matches
|
||||
(→ `pci-bus`); everything else arrives later as bus reports and matches on
|
||||
@@ -293,7 +295,7 @@ the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so
|
||||
uncacheable, physical address exposed), **memory barriers** (`library/device/mmio`'s
|
||||
`memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, imported as the `mmio` module), **fault isolation** (a ring-3 fault kills only the faulting
|
||||
process — `killCurrentProcess` — and the machine keeps running,
|
||||
[resilience](../os-development-guide/resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
[resilience](../os-development/resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
process releases its claims and IRQ/MSI bindings — `releaseAllOwnedBy`,
|
||||
`irq.releaseOwner` — and the device manager respawns the driver with backoff,
|
||||
[device-manager.md](device-manager.md)). What remains:
|
||||
@@ -394,7 +396,7 @@ Claiming and mapping is half of being a danos driver; the other half is the
|
||||
- Build on `service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](../os-development-guide/process-lifecycle.md)).
|
||||
([process-lifecycle.md](../os-development/process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
+1
-1
@@ -160,5 +160,5 @@ serial line names the class received, so the log shows all three arriving on one
|
||||
## See also
|
||||
|
||||
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||
- [syscall.md](../os-development-guide/syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [syscall.md](../os-development/syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||
@@ -18,7 +18,7 @@ There are two layers, built a milestone apart:
|
||||
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](../os-development-guide/scheduling.md).
|
||||
scheduler's [wait queues](../os-development/scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
@@ -37,7 +37,7 @@ Two details make it correct:
|
||||
rather than assuming the slot is still available — another waiter may have taken
|
||||
it first. This is the standard guard against spurious or racing wakeups.
|
||||
- **One critical section.** `send`/`receive` run under the [big kernel
|
||||
lock](../os-development-guide/smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
lock](../os-development/smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
core *and* takes the kernel's one spinlock — since SMP, the interrupt flag alone
|
||||
is not atomicity, because `cli` on one core does nothing to another. So checking
|
||||
the condition and committing the block/enqueue happen atomically both with respect
|
||||
@@ -114,7 +114,7 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](../os-development-guide/process-lifecycle.md) ride the
|
||||
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
||||
notification mechanism:
|
||||
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
@@ -49,7 +49,7 @@ addressed by device id. `/dev` is the much smaller set of devices that have a dr
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](../device-driver-development-guide/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
([drivers.md](../device-driver-development/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
||||
@@ -92,7 +92,7 @@ descriptor ring and left to read and write memory on its own. That ring is exact
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/device/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development-guide/driver-model.md); the earlier
|
||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development/driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
> backend, unchanged. The Zig source of truth is `library/protocol/vfs/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit test pins a sample of the sizes
|
||||
> and values below. This page is the **language-neutral wire specification**
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](../os-development-guide/vdso.md)
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](../os-development/vdso.md)
|
||||
> explains why the IPC protocols, not the syscall numbers, are danos's
|
||||
> public ABI).
|
||||
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
# OS Developer Guide
|
||||
|
||||
This document is for those who need to understand the architectural decisions behind the OS.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The os was initially written in zig because it has excellent support for EFI. With zig, we could forgo using a third party bootloader, reducing the time to boot up the kernel. Following the "Zen of Zig", helped to produce the most readable codebase for an operating system ever created. So those, new to OS development could quickly get up to speed.
|
||||
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
# OS Development
|
||||
|
||||
This document explains the architectural decisions behind the operating system.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The OS is written in Zig because it has excellent EFI support, so the OS boots quickly without a third-party bootloader.
|
||||
|
||||
Zig comes batteries included for systems work — cross-compilation, a build system, and a test runner are all part of the toolchain. Building with `-Doptimize=ReleaseSafe` keeps runtime safety checks on in the shipped kernel, which removes entire classes of bugs. The built-in test suite, combined with a QEMU integration harness, means every feature is proven to work, before it is shipped.
|
||||
|
||||
The codebase of the OS prioritizes readability. The aim is a codebase where someone new to OS development can find their way around without a guide.
|
||||
|
||||
## A microkernel?
|
||||
|
||||
The kernel is a thin layer: it schedules processes and manages memory. Everything else — drivers, file systems, the display — runs in user space as separate, isolated processes.
|
||||
|
||||
The payoff is resilience. When a driver crashes, it doesn't take the OS down with it; it gets restarted. That makes this an ideal environment for *developing* an operating system, because a buggy driver is an ordinary bug: patch it, restart the service, and keep going.
|
||||
|
||||
There is a security benefit too. Processes are isolated and talk over Inter-Process Communication (IPC) channels, so compromising one service doesn't hand an attacker the whole machine. Vulnerabilities tend to stay contained in the process they started in.
|
||||
|
||||
Other operating systems choose to pack all of these duties into one binary as a Monolithic kernel, mostly for performance: a function call inside the kernel is faster than passing a message between isolated processes. That cost is real — an IPC round-trip is a few microseconds where a function call is nanoseconds — but it is also workload-shaped. Compute-bound programs don't notice it at all. For bulk data like file contents and pixels, the design moves data through shared memory and DMA so it is copied once, the same as a monolithic kernel; only small control messages cross the IPC boundary. What remains is the per-message cost on chatty paths, and the scheduler and memory management are designed to keep that small.
|
||||
|
||||
## Private ABI
|
||||
|
||||
The syscall layer is private. The numbers and structures in `abi.zig` are an internal detail shared between the kernel and the system's own libraries, and they are free to change between builds.
|
||||
|
||||
The public boundary sits one level up: the [vDSO](vdso.md) that programs call into, and the documented IPC protocols such as the [VFS protocol](../file-system-development/vfs-protocol.md). Programs that stick to those interfaces keep working while the kernel rearranges itself underneath. This is the opposite of the Linux approach, where raw syscall numbers are frozen forever; here, stability is promised at the library and protocol level, and nowhere below it.
|
||||
|
||||
## Steal the best bits and dump the legacy
|
||||
|
||||
The OS is Unix-like, but selectively. It borrows the ideas that have aged well — everything is a file, small services composed over clean interfaces — and skips the parts of POSIX that have caused decades of headaches.
|
||||
|
||||
Some concrete choices:
|
||||
|
||||
- **`spawn`, not `fork`.** Creating a process starts a fresh program and returns the child's id. There is no clone-the-whole-address-space-then-immediately-throw-it-away dance, and none of the subtle state-inheritance bugs that come with it.
|
||||
- **Time is a syscall.** The kernel owns the clock and timers directly. There is no time daemon to keep alive and no ambiguity about where the truth lives.
|
||||
- **Lifecycle events arrive as messages.** A supervisor learns that a child exited through an IPC message on an endpoint it already owns — delivered like any other message, not as an interrupt that can fire between any two instructions.
|
||||
|
||||
The test for keeping an idea is simple: does it still pull its weight, or is it only there because it was there in 1979?
|
||||
@@ -75,7 +75,7 @@ There are really two independent questions, and it's worth not conflating them:
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables and address-space
|
||||
management (see [paging.md](paging.md)).
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** / **`ioapic.zig`** — the Local APIC, its timer,
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](../device-driver-development-guide/device-interrupts.md)).
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](../device-driver-development/device-interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](../testing.md)) and the shared port-I/O + MSR primitives.
|
||||
- **`system/kernel/architecture/x86_64/smp.zig`** / **`per-cpu.zig`** — application-processor bring-up
|
||||
@@ -132,7 +132,7 @@ when*:
|
||||
- **User-space enumeration: a device-manager server.** Everything else — PCI devices,
|
||||
peripherals — is parsed (or queried from the kernel's parse) by a privileged
|
||||
user-space server that hands each driver process its MMIO regions and IRQ rights
|
||||
over [IPC](../device-driver-development-guide/ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
over [IPC](../device-driver-development/ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
driver as a message on a channel — a natural extension of the wait queues and
|
||||
channels already built), that's what makes drivers genuinely isolated.
|
||||
|
||||
@@ -144,7 +144,7 @@ slice is unavoidably in-kernel.
|
||||
On ARMv8 the generic timer exposes its frequency directly via the `CNTFRQ` register —
|
||||
no calibration needed. That's cleaner than the x86 side, where we measure the LAPIC
|
||||
and TSC against the PIT because nothing tells us their frequency (see
|
||||
[device-interrupts.md](../device-driver-development-guide/device-interrupts.md)). Discovery on ARM hands you more for
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md)). Discovery on ARM hands you more for
|
||||
free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
|
||||
## Suggested ordering
|
||||
@@ -166,9 +166,9 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [arm.md](arm.md) — the aarch64 target that forces genuine discovery (DTB, GIC).
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam,
|
||||
and the note about grabbing the RSDP before `ExitBootServices`.
|
||||
- [device-interrupts.md](../device-driver-development-guide/device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
- [device-interrupts.md](../device-driver-development/device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
discovery will eventually feed (IOAPIC, real IRQ routing).
|
||||
- [ipc.md](../device-driver-development-guide/ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
- [ipc.md](../device-driver-development/ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](../vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
@@ -177,7 +177,7 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](../device-driver-development-guide/device-manager.md)): it claims the bridge, repeats the
|
||||
([device-manager.md](../device-driver-development/device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
@@ -190,7 +190,7 @@ for the host bridge, FADT); at this point it also still built the AML namespace
|
||||
but only to read the `\\_S5` sleep type for poweroff. (That remnant is gone too:
|
||||
the kernel now runs no AML at all — soft-off belongs to the acpi service, and the
|
||||
kernel keeps only the AML-free reboot path.) Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](../device-driver-development-guide/device-manager.md)): it claims the `acpi-tables`
|
||||
service** ([device-manager.md](../device-driver-development/device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
@@ -206,7 +206,7 @@ ring 0.)
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](../device-driver-development-guide/device-manager.md) protocol — descriptors,
|
||||
above the [device-manager](../device-driver-development/device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
@@ -136,7 +136,7 @@ Both items originally deferred here have landed:
|
||||
- **The IO-APIC**: [ioapic.zig](../../system/kernel/architecture/x86_64/ioapic.zig)
|
||||
routes external device lines onto vectors — discovered via ACPI's MADT, every
|
||||
input masked at init, lines unmasked one at a time as user-space drivers bind
|
||||
them (see [device-interrupts.md](../device-driver-development-guide/device-interrupts.md)). The keyboard followed
|
||||
them (see [device-interrupts.md](../device-driver-development/device-interrupts.md)). The keyboard followed
|
||||
exactly as predicted: the PS/2 bus driver (`system/drivers/ps2-bus/`) claims
|
||||
the 8042 controller and binds its IRQ 1 (and the aux mouse's IRQ 12) through
|
||||
this routing. USB HID keyboards arrive over xHCI instead, which interrupts via
|
||||
@@ -138,8 +138,8 @@ Four tests (see [testing.md](../testing.md)) pin down the guarantees:
|
||||
processes own the low half.
|
||||
- **Per-address-space tables** — done: each user process gets its own root with
|
||||
the kernel half shared, and refcounted shared-memory mappings exist
|
||||
([ipc.md](../device-driver-development-guide/ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
([ipc.md](../device-driver-development/ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
- **Uncacheable MMIO** — half done: user-space device and DMA mappings are
|
||||
strong-uncacheable and the framebuffer is write-combining via the PAT, but the
|
||||
kernel's own `mapMmio` path is still writeback — the LAPIC included (see
|
||||
[device-interrupts.md](../device-driver-development-guide/device-interrupts.md)).
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md)).
|
||||
@@ -7,7 +7,7 @@ them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](../device-driver-development-guide/input.md), applied to power.
|
||||
publish/subscribe shape as the [input service](../device-driver-development/input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
@@ -126,5 +126,5 @@ until laptop sleep), and thermal zones.
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](../device-driver-development-guide/device-manager.md) — the supervision model init mirrors for
|
||||
- [device-manager.md](../device-driver-development/device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
+3
-3
@@ -9,7 +9,7 @@ the layer above them — the standard vocabulary a danos process speaks about it
|
||||
life, and the stable `process` interface that carries it. Nothing here is
|
||||
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||
reload, and die the same way. The device manager is simply this design's first
|
||||
serious customer ([device-manager.md](../device-driver-development-guide/device-manager.md)).
|
||||
serious customer ([device-manager.md](../device-driver-development/device-manager.md)).
|
||||
|
||||
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||
@@ -172,7 +172,7 @@ zombie state or privileged snooping:
|
||||
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||
the same way. This covers the *synchronous* case only.
|
||||
3. **The subscribers** — the new piece, and it is the input service's
|
||||
publish/subscribe shape ([input.md](../device-driver-development-guide/input.md)) applied to exits. A stateful
|
||||
publish/subscribe shape ([input.md](../device-driver-development/input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: a filesystem server
|
||||
(FAT today) holds a dead client's open file handles, the input service holds
|
||||
its subscriptions, a future network stack holds its sockets. None of these
|
||||
@@ -320,7 +320,7 @@ get POSIX; danos-native programs never pay for it.
|
||||
`process` grows the interface above; the service harness handles
|
||||
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||
|
||||
[device-manager.md](../device-driver-development-guide/device-manager.md) builds directly on all four.
|
||||
[device-manager.md](../device-driver-development/device-manager.md) builds directly on all four.
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
@@ -5,7 +5,7 @@ isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](../device-driver-development-guide/device-manager.md)): the device manager supervises every
|
||||
([device-manager.md](../device-driver-development/device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
@@ -81,7 +81,7 @@ Detecting and killing is the easy half. The genuinely tricky questions are about
|
||||
- **In-flight IPC**: messages sent to the dead component, or replies its clients are
|
||||
blocked waiting for. The channel has to break cleanly and unblock the waiters with
|
||||
an error rather than hang them forever (a design constraint that reaches back into
|
||||
[ipc.md](../device-driver-development-guide/ipc.md) — channels need a "peer died" outcome).
|
||||
[ipc.md](../device-driver-development/ipc.md) — channels need a "peer died" outcome).
|
||||
- **Clients**: how does a client discover the service it was talking to is gone and
|
||||
has been replaced? Options: capability revocation makes stale handles fail; or a
|
||||
**name server** re-binds clients to the new instance; or clients retry through a
|
||||
@@ -159,7 +159,7 @@ real-time work without owing anyone a timing *guarantee*.
|
||||
- [vision.md](../vision.md) — the goals this serves (learning by doing; resilience over
|
||||
hard real-time).
|
||||
- [scheduling.md](scheduling.md) — preemption, which makes runaway components killable.
|
||||
- [ipc.md](../device-driver-development-guide/ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [ipc.md](../device-driver-development/ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [interrupts.md](interrupts.md) — fault reporting that user mode turns into "kill and
|
||||
restart" instead of "halt".
|
||||
- [smp.md](smp.md) — the real-time-vs-resilience fork, in the SMP context.
|
||||
@@ -37,7 +37,7 @@ down a return address pointing at `task_trampoline` and zeroed callee-saved slot
|
||||
`schedule()` — pick the best task and switch — runs from two places:
|
||||
|
||||
- **`yield()`** — a task voluntarily gives up the CPU.
|
||||
- **`tick()`** — the 1000 Hz [timer](../device-driver-development-guide/device-interrupts.md) preempts the running
|
||||
- **`tick()`** — the 1000 Hz [timer](../device-driver-development/device-interrupts.md) preempts the running
|
||||
task. This is what lets a task that never yields still share the CPU.
|
||||
|
||||
The subtlety in mixing them is the **interrupt flag (IF)**. The rule: `switch_context`
|
||||
@@ -97,7 +97,7 @@ marks the task blocked with a wake deadline and switches away. On every tick the
|
||||
timer wakes any task whose deadline has passed (a bounded scan, so it stays
|
||||
deterministic), which makes it ready again; the scheduler then runs it when its
|
||||
priority comes up. `sleep` measures its deadline on the [calibrated
|
||||
clock](../device-driver-development-guide/device-interrupts.md), so it's real time.
|
||||
clock](../device-driver-development/device-interrupts.md), so it's real time.
|
||||
|
||||
When *every* task is blocked, something still has to run — so there's an **idle
|
||||
task** at the lowest priority that just `hlt`s until the next interrupt (see
|
||||
@@ -111,7 +111,7 @@ The other form of blocking is waiting for an **event** rather than a duration. A
|
||||
the caller on it, `wake(wq)` moves the highest-priority waiter back to ready
|
||||
(preempting if it now outranks the running task). A task links into a wait queue
|
||||
through the same field the ready queues use — it's in exactly one queue at a time.
|
||||
These are the primitives locks, semaphores and [IPC](../device-driver-development-guide/ipc.md) are built on.
|
||||
These are the primitives locks, semaphores and [IPC](../device-driver-development/ipc.md) are built on.
|
||||
|
||||
Blocking safely needs **composable critical sections**. A blanket `cli`/`sti` pair
|
||||
doesn't nest: an IPC channel that `cli`s and then calls `wait` would have `wait`'s
|
||||
+1
-1
@@ -307,4 +307,4 @@ refcount, and no group-kill special case is needed at all.
|
||||
Then update [threading.md](threading.md) (the shared-fate gap note),
|
||||
[process-lifecycle.md](process-lifecycle.md),
|
||||
[process-management.md](process-management.md), and
|
||||
[ipc.md](../device-driver-development-guide/ipc.md)/[drivers.md](../device-driver-development-guide/drivers.md) mentions.
|
||||
[ipc.md](../device-driver-development/ipc.md)/[drivers.md](../device-driver-development/drivers.md) mentions.
|
||||
@@ -270,6 +270,6 @@ next lands.
|
||||
|
||||
- [scheduling.md](scheduling.md) — the single-core scheduler SMP would extend.
|
||||
- [discovery.md](discovery.md) — enumerating cores is a device-discovery problem.
|
||||
- [ipc.md](../device-driver-development-guide/ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [ipc.md](../device-driver-development/ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](../vision.md) — the goals question (real-time vs resilience) this note
|
||||
keeps bumping into.
|
||||
@@ -51,7 +51,7 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
||||
3. **`Yield()`/`Thread_Ctrl()`**
|
||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](../device-driver-development-guide/input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](../device-driver-development/input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
|
||||
* * * * *
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](../device-driver-development-guide/display-v2-plan.md). Read threading.md first for the *why*.
|
||||
[display-v2-plan.md](../device-driver-development/display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
@@ -459,7 +459,7 @@ clean.
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through a [shared-memory](../device-driver-development-guide/display-v2.md)
|
||||
physical-address key so two processes share a futex through a [shared-memory](../device-driver-development/display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
@@ -36,7 +36,7 @@ implementation underneath, not the API above.
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](../device-driver-development-guide/ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
[ipc.md](../device-driver-development/ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
@@ -214,7 +214,7 @@ Keying: threads share an address space, so a **virtual address within that addre
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shared-memory](../device-driver-development-guide/display-v2.md) region later, without changing the API. We start with the
|
||||
[shared-memory](../device-driver-development/display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
@@ -264,7 +264,7 @@ stays single-threaded and lean.
|
||||
**process**, which respawns its threads from a known-good state — restart
|
||||
granularity stays the process. The leader's recorded exit reason carries the fault
|
||||
class even when a worker faulted, so restart policy is unchanged.
|
||||
- **IPC — two consequences threads forced ([ipc.md](../device-driver-development-guide/ipc.md)):**
|
||||
- **IPC — two consequences threads forced ([ipc.md](../device-driver-development/ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
@@ -284,7 +284,7 @@ stays single-threaded and lean.
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](../device-driver-development-guide/display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
[display-v2-plan.md](../device-driver-development/display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
@@ -343,7 +343,7 @@ the later self-hosting lift cheap.
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](../vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](../device-driver-development-guide/ipc.md) — the private ABI and the messaging model
|
||||
- [syscall.md](syscall.md), [ipc.md](../device-driver-development/ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](../zig-self-hosting.md) — the target this bends toward.
|
||||
@@ -7,7 +7,7 @@ Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](../device-driver-development-guide/device-interrupts.md); the scheduler's blocking and
|
||||
are built in [device-interrupts.md](../device-driver-development/device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
@@ -32,11 +32,11 @@ danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on I
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](../device-driver-development-guide/device-interrupts.md).
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](../device-driver-development-guide/drivers.md), as documentation.) The one place
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
@@ -55,7 +55,7 @@ Time and waiting are three entries in the small syscall table ([syscall.md](sysc
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](../device-driver-development-guide/device-manager.md)).
|
||||
[device-manager.md](../device-driver-development/device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
@@ -118,7 +118,7 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
(`system/kernel/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel` off the FAT boot volume, then loads user
|
||||
space: a prebuilt `boot\system.img` capsule
|
||||
([system-image.md](os-development-guide/system-image.md)) when present, otherwise it walks
|
||||
([system-image.md](os-development/system-image.md)) when present, otherwise it walks
|
||||
the volume's `/system` and optional `/test` trees (init included) into the
|
||||
initial ramdisk. The kernel can boot "kernel-only" without either.
|
||||
(`efi.zig:16`, `efi.zig:68`)
|
||||
|
||||
+2
-2
@@ -27,7 +27,7 @@ boot log, memory summary, exception reports — appears on serial as plain text.
|
||||
|
||||
QEMU captures that with `-serial file:serial.log`, giving a machine-readable
|
||||
transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [architecture](os-development-guide/architecture.md) boundary — and adding
|
||||
memory-mapped UART), so it lives behind the [architecture](os-development/architecture.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
@@ -88,7 +88,7 @@ table in `test/qemu_test.py`):
|
||||
| `fault-recovery` | a ring-3 process that faults is killed and reaped while init keeps heartbeating — the OS survives | `DANOS-TEST-RESULT: PASS` |
|
||||
|
||||
The faulting cases don't print a result line — they deliberately raise a CPU
|
||||
exception, and the harness asserts on the [exception report](os-development-guide/interrupts.md) the
|
||||
exception, and the harness asserts on the [exception report](os-development/interrupts.md) the
|
||||
handler prints (which also reaches serial). This reuses the real fault path as the
|
||||
test oracle: if the IDT/TSS weren't wired up, `fault-df` would triple-fault and the
|
||||
marker would never appear.
|
||||
|
||||
@@ -108,7 +108,7 @@ localised (below).
|
||||
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||
|
||||
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](os-development-guide/syscall.md)); the **`runtime`**
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](os-development/syscall.md)); the **`runtime`**
|
||||
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||
|
||||
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||
@@ -165,7 +165,7 @@ What the seam needs, and what danos already provides:
|
||||
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||
| monotonic clock | `clock` syscall | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](os-development-guide/sysv.md)), `runtime.process.Init` | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](os-development/sysv.md)), `runtime.process.Init` | none |
|
||||
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||
@@ -209,7 +209,7 @@ build); point danos's `build.zig`/CI at the resulting binary. Four localised pat
|
||||
plan9/serenity;
|
||||
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||
and `Init`/argv construction it already builds ([sysv.md](os-development-guide/sysv.md));
|
||||
and `Init`/argv construction it already builds ([sysv.md](os-development/sysv.md));
|
||||
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||
@@ -344,9 +344,9 @@ Two current decisions fall out of this roadmap:
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the north star this serves.
|
||||
- [syscall.md](os-development-guide/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](os-development-guide/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](device-driver-development-guide/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
|
||||
@@ -0,0 +1,339 @@
|
||||
Vendor ID,Device ID,Graphics Family,GPU Name
|
||||
0x8086,0x0152,HD Graphics,Xeon E3-1200 v2/3rd Gen Core GT1
|
||||
0x8086,0x0155,HD Graphics,Xeon E3-1200 v2/3rd Gen Core
|
||||
0x8086,0x0156,HD Graphics,Ivy Bridge mobile GT1
|
||||
0x8086,0x0157,HD Graphics,Ivy Bridge mobile GT1
|
||||
0x8086,0x015a,HD Graphics,Xeon E3-1200 v2/Ivy Bridge
|
||||
0x8086,0x0162,HD Graphics,Ivy Bridge GT2 (HD Graphics 4000)
|
||||
0x8086,0x0166,HD Graphics,Ivy Bridge mobile GT2 (HD Graphics 4000)
|
||||
0x8086,0x016a,HD Graphics,Xeon E3-1200 v2/3rd Gen Core GT3
|
||||
0x8086,0x0402,HD Graphics,Xeon E3-1200 v3/4th Gen Core GT1
|
||||
0x8086,0x0406,HD Graphics,Haswell GT1
|
||||
0x8086,0x040a,HD Graphics,Xeon E3-1200 v3 GT1
|
||||
0x8086,0x040b,HD Graphics,Haswell GT1
|
||||
0x8086,0x040e,HD Graphics,Haswell GT1
|
||||
0x8086,0x0412,HD Graphics,Xeon E3-1200 v3/4th Gen Core GT2
|
||||
0x8086,0x0416,HD Graphics,4th Gen Core GT2
|
||||
0x8086,0x041a,HD Graphics,Xeon E3-1200 v3 GT2
|
||||
0x8086,0x041b,HD Graphics,Haswell GT2
|
||||
0x8086,0x041e,HD Graphics,4th Gen Core Family GT2
|
||||
0x8086,0x0422,HD Graphics,Haswell GT3
|
||||
0x8086,0x0426,HD Graphics,Haswell GT3
|
||||
0x8086,0x042a,HD Graphics,Haswell GT3
|
||||
0x8086,0x042b,HD Graphics,Haswell GT3
|
||||
0x8086,0x042e,HD Graphics,Haswell GT3
|
||||
0x8086,0x0a02,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a06,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0a,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0b,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0e,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a12,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a16,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1a,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1b,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1e,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a22,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a26,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2a,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2b,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2e,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0d02,Iris Pro Graphics,Crystal Well GT1
|
||||
0x8086,0x0d06,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0a,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0b,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0e,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d12,Iris Pro Graphics,Crystal Well GT3 (Iris Pro 5200)
|
||||
0x8086,0x0d16,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1b,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1e,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d22,Iris Pro Graphics,Crystal Well (Iris Pro 5200)
|
||||
0x8086,0x0d26,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2b,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2e,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d32,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d36,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d3a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x1602,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x1606,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160a,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160b,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160d,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160e,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x1612,HD Graphics,Broadwell-H GT2 (HD Graphics 5600)
|
||||
0x8086,0x1616,HD Graphics,Broadwell-U GT2 (HD Graphics 5500)
|
||||
0x8086,0x161a,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161b,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161d,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161e,HD Graphics,Broadwell-Y GT2 (HD Graphics 5300)
|
||||
0x8086,0x1622,Iris Pro Graphics,Broadwell-DT/H GT3 (Iris Pro 6200)
|
||||
0x8086,0x1626,HD Graphics,Broadwell-U GT3 (HD Graphics 6000)
|
||||
0x8086,0x162a,Iris Pro Graphics,Broadwell-DT GT3 (Iris Pro P6300)
|
||||
0x8086,0x162b,Iris Graphics,Broadwell-U GT3 (Iris 6100)
|
||||
0x8086,0x162d,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x162e,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1632,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1636,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163a,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163b,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163d,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163e,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1902,HD Graphics,Skylake-S GT1 (HD Graphics 510)
|
||||
0x8086,0x1906,HD Graphics,Skylake-U GT1 (HD Graphics 510)
|
||||
0x8086,0x190a,HD Graphics,Skylake-DT GT1
|
||||
0x8086,0x190b,HD Graphics,Skylake GT1 (HD Graphics 510)
|
||||
0x8086,0x190e,HD Graphics,Skylake GT1
|
||||
0x8086,0x1912,HD Graphics,Skylake-S GT2 (HD Graphics 530)
|
||||
0x8086,0x1913,HD Graphics,Skylake GT2
|
||||
0x8086,0x1915,HD Graphics,Skylake GT2
|
||||
0x8086,0x1916,HD Graphics,Skylake-U GT2 (HD Graphics 520)
|
||||
0x8086,0x1917,HD Graphics,Skylake GT2
|
||||
0x8086,0x191a,HD Graphics,Skylake GT2
|
||||
0x8086,0x191b,HD Graphics,Skylake-H GT2 (HD Graphics 530)
|
||||
0x8086,0x191d,HD Graphics,Skylake-DT/H GT2 (HD Graphics P530)
|
||||
0x8086,0x191e,HD Graphics,Skylake-Y GT2 (HD Graphics 515)
|
||||
0x8086,0x1921,HD Graphics,Skylake GT2 (HD Graphics 520)
|
||||
0x8086,0x1923,HD Graphics,Skylake GT2 (HD Graphics 535)
|
||||
0x8086,0x1926,Iris Graphics,Skylake-U GT3 (Iris Graphics 540)
|
||||
0x8086,0x1927,Iris Graphics,Skylake-U GT3 (Iris Graphics 550)
|
||||
0x8086,0x192a,Iris Graphics,Skylake GT3
|
||||
0x8086,0x192b,Iris Graphics,Skylake GT3 (Iris Graphics 555)
|
||||
0x8086,0x192d,Iris Graphics,Skylake-H GT3 (Iris Graphics P555)
|
||||
0x8086,0x1932,Iris Pro Graphics,Skylake GT4 (Iris Pro 580)
|
||||
0x8086,0x193a,Iris Pro Graphics,Skylake-H GT4 (Iris Pro P580)
|
||||
0x8086,0x193b,Iris Pro Graphics,Skylake-H GT4 (Iris Pro 580)
|
||||
0x8086,0x193d,Iris Pro Graphics,Skylake-H GT4 (Iris Pro P580)
|
||||
0x8086,0x1a84,UHD Graphics,Skylake-DT
|
||||
0x8086,0x1a85,UHD Graphics,Skylake-DT
|
||||
0x8086,0x5902,HD Graphics,Kaby Lake-S GT1 (HD Graphics 610)
|
||||
0x8086,0x5906,HD Graphics,Kaby Lake-U GT1 (HD Graphics 610)
|
||||
0x8086,0x5908,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x590a,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x590b,HD Graphics,Kaby Lake GT1 (HD Graphics 610)
|
||||
0x8086,0x590e,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x5912,HD Graphics,Kaby Lake-S GT2 (HD Graphics 630)
|
||||
0x8086,0x5913,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x5915,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x5916,HD Graphics,Kaby Lake-U GT2 (HD Graphics 620)
|
||||
0x8086,0x5917,UHD Graphics,Kaby Lake-R GT2 (UHD Graphics 620)
|
||||
0x8086,0x591a,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x591b,HD Graphics,Kaby Lake-H GT2 (HD Graphics 630)
|
||||
0x8086,0x591c,UHD Graphics,Kaby Lake GT2 (UHD Graphics 615)
|
||||
0x8086,0x591d,HD Graphics,Kaby Lake-DT GT2 (HD Graphics P630)
|
||||
0x8086,0x591e,HD Graphics,Kaby Lake-Y GT2 (HD Graphics 615)
|
||||
0x8086,0x5921,HD Graphics,Kaby Lake GT2 (HD Graphics 620)
|
||||
0x8086,0x5923,HD Graphics,Kaby Lake GT2 (HD Graphics 635)
|
||||
0x8086,0x5926,Iris Plus Graphics,Kaby Lake-U GT3 (Iris Plus 640)
|
||||
0x8086,0x5927,Iris Plus Graphics,Kaby Lake-U GT3 (Iris Plus 650)
|
||||
0x8086,0x592a,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x592b,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x5932,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593a,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593b,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593d,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x5a40,Intel Graphics,Apollolake
|
||||
0x8086,0x5a41,Intel Graphics,Apollolake
|
||||
0x8086,0x5a42,Intel Graphics,Apollolake
|
||||
0x8086,0x5a44,Intel Graphics,Apollolake
|
||||
0x8086,0x5a49,Intel Graphics,Apollolake
|
||||
0x8086,0x5a4a,Intel Graphics,Apollolake
|
||||
0x8086,0x5a4c,Intel Graphics,Apollolake
|
||||
0x8086,0x5a50,Intel Graphics,Apollolake
|
||||
0x8086,0x5a51,Intel Graphics,Apollolake
|
||||
0x8086,0x5a52,Intel Graphics,Apollolake
|
||||
0x8086,0x5a54,Intel Graphics,Apollolake
|
||||
0x8086,0x5a59,Intel Graphics,Apollolake
|
||||
0x8086,0x5a5a,Intel Graphics,Apollolake
|
||||
0x8086,0x5a5c,Intel Graphics,Apollolake
|
||||
0x8086,0x5a71,Intel Graphics,Apollolake
|
||||
0x8086,0x5a79,Intel Graphics,Apollolake
|
||||
0x8086,0x5a84,HD Graphics,Apollo Lake GT1 (HD Graphics 505)
|
||||
0x8086,0x5a85,HD Graphics,Apollo Lake GT1 (HD Graphics 500)
|
||||
0x8086,0x3184,UHD Graphics,GeminiLake (UHD Graphics 605)
|
||||
0x8086,0x3185,UHD Graphics,GeminiLake (UHD Graphics 600)
|
||||
0x8086,0x3e90,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3e91,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e92,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e93,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3e94,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e96,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e98,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e99,UHD Graphics,Coffee Lake GT2
|
||||
0x8086,0x3e9a,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e9b,UHD Graphics,Coffee Lake-H GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e9c,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3ea0,UHD Graphics,Whiskey Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x3ea1,UHD Graphics,Whiskey Lake-U GT1 (UHD Graphics 610)
|
||||
0x8086,0x3ea2,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea3,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea4,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea5,Iris Plus Graphics,Coffee Lake-U GT3e (Iris Plus 655)
|
||||
0x8086,0x3ea6,Iris Plus Graphics,Coffee Lake-U GT3 (Iris Plus 645)
|
||||
0x8086,0x3ea7,Iris Plus Graphics,Whiskey Lake GT3
|
||||
0x8086,0x3ea8,Iris Plus Graphics,Coffee Lake-U GT3 (Iris Plus 655)
|
||||
0x8086,0x3ea9,UHD Graphics,Coffee Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x87c0,UHD Graphics,9th Gen Core (UHD Graphics 617)
|
||||
0x8086,0x87ca,UHD Graphics,9th Gen Core (UHD Graphics 617)
|
||||
0x8086,0x8a50,Iris Plus Graphics,Ice Lake Gen1
|
||||
0x8086,0x8a51,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a52,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a53,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a54,Iris Plus Graphics,Ice Lake GT2
|
||||
0x8086,0x8a56,Iris Plus Graphics,Ice Lake GT1 (Iris Plus G1)
|
||||
0x8086,0x8a57,Iris Plus Graphics,Ice Lake GT1
|
||||
0x8086,0x8a58,UHD Graphics,Ice Lake-Y GT1 (UHD Graphics G1)
|
||||
0x8086,0x8a59,UHD Graphics,Ice Lake GT1
|
||||
0x8086,0x8a5a,Iris Plus Graphics,Ice Lake GT4 (Iris Plus G4)
|
||||
0x8086,0x8a5b,Iris Plus Graphics,Ice Lake GT4
|
||||
0x8086,0x8a5c,Iris Plus Graphics,Ice Lake GT4 (Iris Plus G4)
|
||||
0x8086,0x8a5d,Iris Plus Graphics,Ice Lake GT4
|
||||
0x8086,0x8a70,Iris Plus Graphics,Ice Lake
|
||||
0x8086,0x8a71,Iris Plus Graphics,Ice Lake
|
||||
0x8086,0x9a40,Iris Xe Graphics,Tiger Lake-UP4 GT2
|
||||
0x8086,0x9a49,Iris Xe Graphics,Tiger Lake-LP GT2
|
||||
0x8086,0x9a59,Iris Xe Graphics,Tiger Lake GT2
|
||||
0x8086,0x9a60,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a68,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a70,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a78,UHD Graphics,Tiger Lake-LP GT2 (UHD Graphics G4)
|
||||
0x8086,0x9ac0,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9ac9,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9ad9,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9af8,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9b21,UHD Graphics,Comet Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x9b41,UHD Graphics,Comet Lake-U GT2
|
||||
0x8086,0x9ba0,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba2,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba4,UHD Graphics,Comet Lake-H GT1 (UHD Graphics 610)
|
||||
0x8086,0x9ba5,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba8,UHD Graphics,Comet Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x9baa,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bab,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bac,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bc0,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bc2,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bc4,UHD Graphics,Comet Lake-H GT2
|
||||
0x8086,0x9bc5,UHD Graphics,Comet Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x9bc6,UHD Graphics,Comet Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x9bc8,UHD Graphics,Comet Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x9bca,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bcb,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bcc,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9be6,UHD Graphics,Comet Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x9bf6,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x4555,UHD Graphics,Elkhart Lake GT2 (UHD Graphics Gen11 16EU)
|
||||
0x8086,0x4571,UHD Graphics,Elkhart Lake GT2 (UHD Graphics Gen11 32EU)
|
||||
0x8086,0x4500,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4541,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4551,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4557,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4570,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4c80,Intel Graphics,Rocket Lake
|
||||
0x8086,0x4c8a,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics 750)
|
||||
0x8086,0x4c8b,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4c8c,Intel Graphics,Rocket Lake GT1
|
||||
0x8086,0x4c90,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics P750)
|
||||
0x8086,0x4c9a,UHD Graphics,Rocket Lake-S
|
||||
0x8086,0x4e51,Intel Graphics,Jasper Lake GT1
|
||||
0x8086,0x4e55,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4e57,Intel Graphics,Jasper Lake GT1
|
||||
0x8086,0x4e61,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4e71,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4626,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x4628,UHD Graphics,Alder Lake-UP3 GT2
|
||||
0x8086,0x462a,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x4680,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0x4681,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4682,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4683,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4688,UHD Graphics,Alder Lake-HX GT1 (UHD Graphics 770)
|
||||
0x8086,0x4689,Intel Graphics,Alder Lake-HX GT1
|
||||
0x8086,0x468a,Intel Graphics,Alder Lake-S
|
||||
0x8086,0x468b,Intel Graphics,Alder Lake-S
|
||||
0x8086,0x4690,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0x4691,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4692,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4693,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 710)
|
||||
0x8086,0x46a0,Intel Graphics,Alder Lake-P GT2
|
||||
0x8086,0x46a1,UHD Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a2,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a3,UHD Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a6,Iris Xe Graphics,Alder Lake-P GT2
|
||||
0x8086,0x46a8,Iris Xe Graphics,Alder Lake-UP3 GT2
|
||||
0x8086,0x46aa,Iris Xe Graphics,Alder Lake-UP4 GT2
|
||||
0x8086,0x46b0,Iris Xe Graphics,Alder Lake-P
|
||||
0x8086,0x46b1,Iris Xe Graphics,Alder Lake-P
|
||||
0x8086,0x46b2,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46b3,UHD Graphics,Alder Lake-UP3 GT1
|
||||
0x8086,0x46c0,Intel Graphics,Alder Lake-M GT1
|
||||
0x8086,0x46c1,Iris Xe Graphics,Alder Lake-M
|
||||
0x8086,0x46c2,Intel Graphics,Alder Lake-M GT1
|
||||
0x8086,0x46c3,UHD Graphics,Alder Lake-UP4 GT1
|
||||
0x8086,0x46d0,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d1,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d2,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d3,Intel Graphics,Alder Lake-N
|
||||
0x8086,0x46d4,Intel Graphics,Alder Lake-N
|
||||
0x8086,0x7d40,Intel Graphics,Meteor Lake-M
|
||||
0x8086,0x7d41,Intel Graphics,Arrow Lake-U
|
||||
0x8086,0x7d45,Intel Graphics,Meteor Lake-P
|
||||
0x8086,0x7d51,Arc Pro Graphics,Arrow Lake-P (Arc Pro 130T/140T)
|
||||
0x8086,0x7d55,Intel Arc Graphics,Meteor Lake-P
|
||||
0x8086,0x7d60,Intel Graphics,Meteor Lake-M
|
||||
0x8086,0x7d67,Intel Graphics,Arrow Lake-S
|
||||
0x8086,0x7dd1,Intel Graphics,Arrow Lake-P
|
||||
0x8086,0x7dd5,Intel Graphics,Meteor Lake-P
|
||||
0x8086,0xa720,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa721,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa780,UHD Graphics,Raptor Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0xa781,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa782,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa783,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa788,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa789,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa78a,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa78b,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa7a0,Iris Xe Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a1,Iris Xe Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a8,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a9,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa7aa,Intel Graphics,Raptor Lake-P
|
||||
0x8086,0xa7ab,Intel Graphics,Raptor Lake-P
|
||||
0x8086,0xa7ac,Intel Graphics,Raptor Lake-U
|
||||
0x8086,0xa7ad,Intel Graphics,Raptor Lake-U
|
||||
0x8086,0xb640,Intel Graphics,Arrow Lake-H
|
||||
0x8086,0x4905,Iris Xe MAX Graphics,DG1
|
||||
0x8086,0x4906,Iris Xe Graphics,DG1
|
||||
0x8086,0x4907,Intel Graphics,DG1 Server
|
||||
0x8086,0x4908,Iris Xe Graphics,DG1
|
||||
0x8086,0x4909,Iris Xe MAX Graphics,DG1 (Iris Xe MAX 100)
|
||||
0x8086,0x5690,Arc Graphics,DG2 (Arc A770M)
|
||||
0x8086,0x5691,Arc Graphics,DG2 (Arc A730M)
|
||||
0x8086,0x5692,Arc Graphics,DG2 (Arc A550M)
|
||||
0x8086,0x5693,Arc Graphics,DG2 (Arc A370M)
|
||||
0x8086,0x5694,Arc Graphics,DG2 (Arc A350M)
|
||||
0x8086,0x5695,Iris Xe MAX Graphics,DG2 (Iris Xe MAX A200M)
|
||||
0x8086,0x5696,Arc Graphics,DG2 (Arc A570M)
|
||||
0x8086,0x5697,Arc Graphics,DG2 (Arc A530M)
|
||||
0x8086,0x5698,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a0,Arc Graphics,DG2 (Arc A770)
|
||||
0x8086,0x56a1,Arc Graphics,DG2 (Arc A750)
|
||||
0x8086,0x56a2,Arc Graphics,DG2 (Arc A580)
|
||||
0x8086,0x56a3,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a4,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a5,Arc Graphics,DG2 (Arc A380)
|
||||
0x8086,0x56a6,Arc Graphics,DG2 (Arc A310)
|
||||
0x8086,0x56b0,Arc Pro Graphics,DG2 (Arc Pro A30M)
|
||||
0x8086,0x56b1,Arc Pro Graphics,DG2 (Arc Pro A40/A50)
|
||||
0x8086,0x56b2,Arc Pro Graphics,DG2 (Arc Pro A60M)
|
||||
0x8086,0x56b3,Arc Pro Graphics,DG2 (Arc Pro A60)
|
||||
0x8086,0x56ba,Arc Graphics,DG2 (Arc A380E)
|
||||
0x8086,0x56bb,Arc Graphics,DG2 (Arc A310E)
|
||||
0x8086,0x56bc,Arc Graphics,DG2 (Arc A370E)
|
||||
0x8086,0x56bd,Arc Graphics,DG2 (Arc A350E)
|
||||
0x8086,0x56be,Arc Graphics,DG2 (Arc A750E)
|
||||
0x8086,0x56bf,Arc Graphics,DG2 (Arc A580E)
|
||||
0x8086,0x56c0,Arc Graphics,DG2 (Data Center GPU Flex 170)
|
||||
0x8086,0x56c1,Arc Graphics,DG2 (Data Center GPU Flex 140)
|
||||
0x8086,0x56c2,Arc Graphics,DG2 (Data Center GPU Flex 170V)
|
||||
|
@@ -0,0 +1,34 @@
|
||||
# /etc/devices.csv — the device→driver registry.
|
||||
#
|
||||
# The device manager reads this at boot and binds each device a bus driver
|
||||
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
||||
# matches goes unbound (logged), never guessed. Edit this file to teach the
|
||||
# system new hardware — no recompile of the device manager required.
|
||||
#
|
||||
# One rule per line, nine comma-separated fields. '#' starts a comment
|
||||
# (whole-line or trailing); blank lines are ignored. Whitespace around a field
|
||||
# is trimmed, so columns may be padded for readability.
|
||||
#
|
||||
# bus which bus reported the device: pci | usb | acpi
|
||||
# base PCI base class / USB class (hex)
|
||||
# class PCI subclass / USB subclass (hex)
|
||||
# prog_if PCI prog-IF / USB protocol (hex)
|
||||
# vendor PCI vendor id / USB idVendor (hex)
|
||||
# device PCI device id / USB idProduct (hex)
|
||||
# subsystem PCI subsystem, packed (ssvid<<16)|ssid (hex)
|
||||
# hid ACPI _HID string (e.g. PNP0303); blank for pci/usb
|
||||
# driver full ramdisk path of the driver to spawn
|
||||
#
|
||||
# '*' or an empty field is a wildcard. When several rows match one device the
|
||||
# MOST SPECIFIC wins (pinning vendor/device/hid beats pinning only a class), so
|
||||
# a generic class rule and a precise vendor:device rule can coexist.
|
||||
#
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
usb, 03, 01, 02, *, *, *, *, /system/drivers/usb-hid-mouse
|
||||
usb, 08, 06, 50, *, *, *, *, /system/drivers/usb-storage
|
||||
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||
|
@@ -0,0 +1,12 @@
|
||||
# /etc/init.csv — diagnose variant (-Ddiagnose), bundled at /etc/init.csv.
|
||||
#
|
||||
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
||||
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
||||
# storage, logger) stays readable on real hardware with no serial. See etc/init.csv
|
||||
# for the format; this file must otherwise track it.
|
||||
#
|
||||
# service args...
|
||||
/system/services/input
|
||||
/system/services/device-manager
|
||||
/system/services/fat
|
||||
/system/services/logger
|
||||
|
@@ -0,0 +1,20 @@
|
||||
# /etc/init.csv — the services init (PID 1) starts at boot, in order.
|
||||
#
|
||||
# init reads this at startup and spawns each service supervised (restarting it on
|
||||
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
||||
# the logger (last) goes down first and its final drain still has the fat server
|
||||
# and the whole storage chain alive underneath it. It is AUTHORITATIVE — there is
|
||||
# no hardcoded fallback list; a missing file means no services are started.
|
||||
#
|
||||
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
||||
# first field is the service binary path; any fields after it are the service's
|
||||
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
||||
# spawns those (see /etc/devices.csv).
|
||||
#
|
||||
# service args...
|
||||
/system/services/input
|
||||
/system/services/device-manager
|
||||
/system/services/fat
|
||||
/system/services/display
|
||||
/system/services/display-demo
|
||||
/system/services/logger
|
||||
|
@@ -0,0 +1,56 @@
|
||||
//! Minimal CSV helpers shared by the `/etc/*.csv` config files — the device
|
||||
//! registry (`/etc/devices.csv`) and the init service list (`/etc/init.csv`).
|
||||
//! Freestanding, no allocator: returned fields are slices into the source line,
|
||||
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
||||
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Strip a trailing `#` comment and surrounding whitespace from one raw line.
|
||||
/// A blank or comment-only line returns "" (length 0) — the caller's skip signal.
|
||||
pub fn stripComment(raw: []const u8) []const u8 {
|
||||
const body = if (std.mem.indexOfScalar(u8, raw, '#')) |hash| raw[0..hash] else raw;
|
||||
return std.mem.trim(u8, body, " \t\r\n");
|
||||
}
|
||||
|
||||
/// Iterate the comma-separated fields of a line body, each trimmed of spaces and
|
||||
/// tabs. Build it from a `stripComment`ed body.
|
||||
pub const Fields = struct {
|
||||
inner: std.mem.SplitIterator(u8, .scalar),
|
||||
|
||||
/// The next field, trimmed, or null when the row is exhausted.
|
||||
pub fn next(self: *Fields) ?[]const u8 {
|
||||
const field = self.inner.next() orelse return null;
|
||||
return std.mem.trim(u8, field, " \t");
|
||||
}
|
||||
};
|
||||
|
||||
pub fn fields(body: []const u8) Fields {
|
||||
return .{ .inner = std.mem.splitScalar(u8, body, ',') };
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
test "stripComment trims and drops comments" {
|
||||
try testing.expectEqualStrings("a, b", stripComment(" a, b # trailing\r\n"));
|
||||
try testing.expectEqualStrings("", stripComment(" # whole-line comment"));
|
||||
try testing.expectEqualStrings("", stripComment(" \t "));
|
||||
try testing.expectEqualStrings("x", stripComment("x"));
|
||||
}
|
||||
|
||||
test "fields splits and trims each column" {
|
||||
var it = fields(stripComment("pci, 03 , 80 , /system/drivers/x # note"));
|
||||
try testing.expectEqualStrings("pci", it.next().?);
|
||||
try testing.expectEqualStrings("03", it.next().?);
|
||||
try testing.expectEqualStrings("80", it.next().?);
|
||||
try testing.expectEqualStrings("/system/drivers/x", it.next().?);
|
||||
try testing.expect(it.next() == null);
|
||||
}
|
||||
|
||||
test "a single field yields one column then null" {
|
||||
var it = fields(stripComment("/system/services/input"));
|
||||
try testing.expectEqualStrings("/system/services/input", it.next().?);
|
||||
try testing.expect(it.next() == null);
|
||||
}
|
||||
@@ -28,6 +28,18 @@ pub const Device = struct {
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Hand the block server a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc) so it forwards it to the controller and the buffer's physical
|
||||
/// addresses become reachable by the device. Call once per buffer before naming it
|
||||
/// in `read`/`write`. Harmless success when no IOMMU is enforcing.
|
||||
pub fn attach(self: Device, handle: ipc.Handle) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.attach), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(self.endpoint, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
|
||||
@@ -99,6 +99,27 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Map a delegated DMA-region (or shared-memory) capability into a claimed device's
|
||||
/// IOMMU domain, so the device may DMA to that buffer. The caller must own `device_id`
|
||||
/// and hold `handle` (received over IPC or from its own `dma.alloc(.. | shareable)`).
|
||||
/// Idempotent. Returns true on success (and trivially when no IOMMU is present).
|
||||
pub fn dmaBind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_bind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Unmap a previously `dmaBind`'d buffer from the device's domain.
|
||||
pub fn dmaUnbind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_unbind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
|
||||
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
|
||||
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
|
||||
/// 0 when no IOMMU is present.
|
||||
pub fn iommuFaultDrain() usize {
|
||||
return sc.systemCall0(.iommu_fault_drain);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
|
||||
@@ -125,6 +125,15 @@ pub const DeviceDescriptor = extern struct {
|
||||
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||
// names with the pci-class module.
|
||||
pci_class: u64,
|
||||
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||
// ChildAdded so /etc/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||
// out identically until they choose to set them.
|
||||
vendor: u16 = 0,
|
||||
device: u16 = 0,
|
||||
subsystem: u32 = 0,
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
|
||||
@@ -52,16 +52,134 @@ pub const config_vendor_id: usize = 0x00;
|
||||
pub const config_device_id: usize = 0x02;
|
||||
pub const config_command: usize = 0x04;
|
||||
pub const config_status: usize = 0x06;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_revision_id: usize = 0x08;
|
||||
pub const config_class_code: usize = 0x09; // 3 bytes: prog-IF 0x09, subclass 0x0A, base class 0x0B
|
||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||
pub const config_subsystem_vendor_id: usize = 0x2C;
|
||||
pub const config_subsystem_id: usize = 0x2E;
|
||||
pub const config_expansion_rom: usize = 0x30;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_interrupt_line: usize = 0x3C;
|
||||
pub const config_interrupt_pin: usize = 0x3D; // 0 = none, 1..4 = INTA..INTD
|
||||
|
||||
/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2).
|
||||
pub const command_memory_and_bus_master: u16 = 0x06;
|
||||
/// Command register bits.
|
||||
pub const command_io_space: u16 = 0x0001; // bit 0: I/O-space decode enable
|
||||
pub const command_memory_space: u16 = 0x0002; // bit 1: memory-space decode enable
|
||||
pub const command_bus_master: u16 = 0x0004; // bit 2: bus-master (DMA) enable
|
||||
pub const command_interrupt_disable: u16 = 0x0400; // bit 10: suppress legacy INTx (MSI/MSI-X unaffected)
|
||||
/// The pair a bus-mastering driver enables together: decode my BARs, let me DMA.
|
||||
pub const command_memory_and_bus_master: u16 = command_memory_space | command_bus_master;
|
||||
|
||||
/// Status register bit 3: legacy INTx is asserted (upstream of the command bit-10 gate).
|
||||
pub const status_interrupt: u16 = 0x0008;
|
||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||
pub const status_capabilities_list: u16 = 0x10;
|
||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||
pub const capability_pointer_mask: u8 = 0xFC;
|
||||
|
||||
/// Capability IDs — the first byte of each entry in the legacy capability list.
|
||||
/// Non-exhaustive: hardware may report IDs not named here.
|
||||
pub const CapabilityId = enum(u8) {
|
||||
power_management = 0x01,
|
||||
msi = 0x05,
|
||||
vendor_specific = 0x09,
|
||||
pci_express = 0x10,
|
||||
msix = 0x11,
|
||||
_,
|
||||
};
|
||||
|
||||
/// MSI capability (id 0x05) register layout. Offsets are relative to the capability
|
||||
/// header; whether the address is one or two dwords (and therefore where the data word
|
||||
/// sits) depends on `control_64bit_capable`.
|
||||
pub const msi = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_enable: u16 = 0x0001;
|
||||
pub const control_multiple_message_capable_mask: u16 = 0x000E; // bits 3:1, log2(vectors requested)
|
||||
pub const control_multiple_message_enable_mask: u16 = 0x0070; // bits 6:4, log2(vectors granted)
|
||||
pub const control_64bit_capable: u16 = 0x0080; // bit 7: address is 64-bit (layout shifts)
|
||||
pub const control_per_vector_masking: u16 = 0x0100; // bit 8
|
||||
pub const address: usize = 0x04; // u32 low address dword (both layouts)
|
||||
pub const address_high: usize = 0x08; // u32, present only when 64-bit capable
|
||||
pub const data_32: usize = 0x08; // u16 message data, 32-bit layout
|
||||
pub const data_64: usize = 0x0C; // u16 message data, 64-bit layout
|
||||
pub const mask_bits_32: usize = 0x0C; // u32, only with per-vector masking
|
||||
pub const mask_bits_64: usize = 0x10;
|
||||
};
|
||||
|
||||
/// MSI-X capability (id 0x11) register layout, plus the 16-byte vector table entry that
|
||||
/// lives in BAR space (not configuration space) at the decoded (BIR, offset).
|
||||
pub const msix = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_table_size_mask: u16 = 0x07FF; // bits 10:0, encoded as N-1
|
||||
pub const control_function_mask: u16 = 0x4000; // bit 14: mask every vector
|
||||
pub const control_enable: u16 = 0x8000; // bit 15
|
||||
pub const table_offset_word: usize = 0x04; // u32: BIR in bits 2:0, table offset in bits 31:3
|
||||
pub const pba_offset_word: usize = 0x08; // u32: same encoding, pending-bit array
|
||||
pub const bir_mask: u32 = 0x0000_0007;
|
||||
pub const offset_mask: u32 = 0xFFFF_FFF8;
|
||||
pub const entry_size: usize = 16; // table entry stride; offsets within an entry:
|
||||
pub const entry_address: usize = 0x0; // u32 low
|
||||
pub const entry_address_high: usize = 0x4; // u32 high
|
||||
pub const entry_data: usize = 0x8; // u32
|
||||
pub const entry_vector_control: usize = 0xC; // u32
|
||||
pub const entry_vector_control_masked: u32 = 0x1; // bit 0; entries reset to masked
|
||||
|
||||
/// Where the table (or pending-bit array) lives, decoded from its offset/BIR dword.
|
||||
pub const TableLocation = struct { bar: u8, offset: u32 };
|
||||
pub fn tableLocation(word: u32) TableLocation {
|
||||
return .{ .bar = @intCast(word & bir_mask), .offset = word & offset_mask };
|
||||
}
|
||||
/// Number of table entries (the control field encodes N-1).
|
||||
pub fn tableSize(control_value: u16) u16 {
|
||||
return (control_value & control_table_size_mask) + 1;
|
||||
}
|
||||
};
|
||||
|
||||
/// Power-management capability (id 0x01) register layout.
|
||||
pub const power_management = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PMC (read-only: version, D-state support)
|
||||
pub const control_status: usize = 0x04; // u16 PMCSR
|
||||
pub const control_status_power_state_mask: u16 = 0x0003; // bits 1:0
|
||||
pub const power_state_d0: u16 = 0x0;
|
||||
pub const power_state_d3_hot: u16 = 0x3;
|
||||
pub const control_status_pme_enable: u16 = 0x0100; // bit 8: plain RW — preserve on writes
|
||||
pub const control_status_pme_status: u16 = 0x8000; // bit 15: RW1C — write 0 or you clear it
|
||||
};
|
||||
|
||||
/// PCI Express capability (id 0x10) register layout — the slice function-level reset
|
||||
/// needs; the full capability is much larger.
|
||||
pub const pci_express = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PCIe Capabilities register
|
||||
pub const device_capabilities: usize = 0x04; // u32
|
||||
pub const device_capabilities_flr: u32 = 1 << 28; // Function Level Reset supported
|
||||
pub const device_control: usize = 0x08; // u16
|
||||
pub const device_control_initiate_flr: u16 = 1 << 15;
|
||||
pub const device_status: usize = 0x0A; // u16
|
||||
pub const device_status_transactions_pending: u16 = 1 << 5;
|
||||
};
|
||||
|
||||
/// Extended (PCI Express) capabilities start here in the 4 KiB configuration space; a
|
||||
/// conventional-PCI function has nothing there (the space reads as all-ones).
|
||||
pub const extended_capability_start: usize = 0x100;
|
||||
/// Extended-capability next pointers are dword-aligned within the 4 KiB space.
|
||||
pub const extended_capability_pointer_mask: u16 = 0xFFC;
|
||||
|
||||
/// The 32-bit header at the start of each extended capability: ID in bits 15:0,
|
||||
/// version in 19:16, next offset in 31:20 (0 = end of list).
|
||||
pub const ExtendedCapabilityHeader = struct {
|
||||
id: u16,
|
||||
version: u4,
|
||||
next: u16,
|
||||
|
||||
pub fn decode(word: u32) ExtendedCapabilityHeader {
|
||||
return .{
|
||||
.id = @truncate(word),
|
||||
.version = @truncate(word >> 16),
|
||||
.next = @intCast((word >> 20) & extended_capability_pointer_mask),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||
/// the dword with the low 4 flag bits masked off.
|
||||
@@ -595,3 +713,37 @@ test "named parts pack to the raw triple" {
|
||||
};
|
||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||
}
|
||||
|
||||
test "MSI-X table word decodes to BIR and offset" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// BIR 3, table at 0x2000 within that BAR.
|
||||
try eq(msix.TableLocation{ .bar = 3, .offset = 0x2000 }, msix.tableLocation(0x0000_2003));
|
||||
// BIR 0, offset 0 — the degenerate-but-common "table at BAR start" case.
|
||||
try eq(msix.TableLocation{ .bar = 0, .offset = 0 }, msix.tableLocation(0));
|
||||
// Table size encodes N-1 in bits 10:0; enable/function-mask bits must not leak in.
|
||||
try eq(@as(u16, 11), msix.tableSize(msix.control_enable | 0x000A));
|
||||
try eq(@as(u16, 1), msix.tableSize(0));
|
||||
try eq(@as(u16, 2048), msix.tableSize(msix.control_table_size_mask));
|
||||
}
|
||||
|
||||
test "extended capability header unpacks id, version, next" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// AER (id 0x0001), version 1, next capability at 0x140.
|
||||
const aer = ExtendedCapabilityHeader.decode(0x1401_0001);
|
||||
try eq(@as(u16, 0x0001), aer.id);
|
||||
try eq(@as(u4, 1), aer.version);
|
||||
try eq(@as(u16, 0x140), aer.next);
|
||||
// A zero header is the "nothing here" terminator.
|
||||
const none = ExtendedCapabilityHeader.decode(0);
|
||||
try eq(@as(u16, 0), none.id);
|
||||
try eq(@as(u16, 0), none.next);
|
||||
}
|
||||
|
||||
test "command bits and capability ids compose" {
|
||||
const eq = std.testing.expectEqual;
|
||||
try eq(command_memory_space | command_bus_master, command_memory_and_bus_master);
|
||||
try eq(@as(u8, 0x05), @intFromEnum(CapabilityId.msi));
|
||||
try eq(@as(u8, 0x11), @intFromEnum(CapabilityId.msix));
|
||||
try eq(@as(u8, 0x01), @intFromEnum(CapabilityId.power_management));
|
||||
try eq(@as(u8, 0x10), @intFromEnum(CapabilityId.pci_express));
|
||||
}
|
||||
|
||||
+283
-7
@@ -1,7 +1,8 @@
|
||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||
//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR
|
||||
//! decode + map, and a capability-list iterator, so a driver never re-derives the
|
||||
//! config-space layout by hand.
|
||||
//! claimed. Config space is mapped as resource 0 (a full 4 KiB ECAM page); this gives
|
||||
//! header-field accessors, BAR decode + map, capability walks (legacy and extended),
|
||||
//! MSI/MSI-X programming, power-state handling, and function-level reset, so a driver
|
||||
//! never re-derives the config-space layout by hand.
|
||||
//!
|
||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||
@@ -13,6 +14,18 @@ const std = @import("std");
|
||||
const mmio = @import("mmio");
|
||||
const pci_class = @import("pci-class");
|
||||
const device = @import("driver");
|
||||
const time = @import("time");
|
||||
|
||||
/// Spec recovery time after a D3hot -> D0 transition.
|
||||
const d0_recovery_millis: u64 = 10;
|
||||
/// How long to wait for in-flight transactions to drain before a function-level reset
|
||||
/// (then reset anyway — resetting a stuck function is the point of FLR).
|
||||
const flr_pending_timeout_millis: u64 = 100;
|
||||
/// The spec's maximum FLR completion time.
|
||||
const flr_settle_millis: u64 = 100;
|
||||
/// How long to wait for the function to become readable again after an FLR.
|
||||
const flr_ready_timeout_millis: u64 = 1000;
|
||||
const flr_poll_interval_millis: u64 = 10;
|
||||
|
||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||
@@ -42,12 +55,61 @@ pub const Function = struct {
|
||||
pub fn status(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||
}
|
||||
pub fn revisionId(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_revision_id);
|
||||
}
|
||||
/// Subsystem vendor ID (config 0x2C) — with `subsystemId`, the standard key for
|
||||
/// board-level quirk matching.
|
||||
pub fn subsystemVendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_vendor_id);
|
||||
}
|
||||
pub fn subsystemId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_id);
|
||||
}
|
||||
/// Interrupt pin (config 0x3D): 0 = none, 1..4 = INTA..INTD.
|
||||
pub fn interruptPin(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_interrupt_pin);
|
||||
}
|
||||
/// The live class-code triple (config 0x09..0x0B), same shape discovery records.
|
||||
pub fn classCode(self: *const Function) pci_class.ClassCode {
|
||||
return .{
|
||||
.prog_if = mmio.readRegister(u8, self.config + pci_class.config_class_code),
|
||||
.subclass = mmio.readRegister(u8, self.config + pci_class.config_class_code + 1),
|
||||
.base = mmio.readRegister(u8, self.config + pci_class.config_class_code + 2),
|
||||
};
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves
|
||||
/// a secondary display's decode off; a bus-mastering device must enable both.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
fn commandSetBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master);
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | bits);
|
||||
}
|
||||
fn commandClearBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) & ~bits);
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master Enable in the command register. Firmware only enables
|
||||
/// memory decode on devices it used at boot; any other device has dead BARs until its
|
||||
/// driver sets it. Bus mastering is separately required for the device to do DMA.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_memory_and_bus_master);
|
||||
}
|
||||
|
||||
/// Clear Bus-Master Enable — stop the device initiating DMA. The quiesce half of a
|
||||
/// driver's shutdown (or a supervisor restart): after this the device can no longer
|
||||
/// write memory the process is about to stop owning.
|
||||
pub fn disableBusMaster(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_bus_master);
|
||||
}
|
||||
|
||||
/// Set command bit 10: suppress legacy INTx assertion. MSI/MSI-X are unaffected —
|
||||
/// set this when enabling either, so the device cannot also raise the shared pin.
|
||||
pub fn setInterruptDisable(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
/// Clear command bit 10, re-allowing legacy INTx assertion.
|
||||
pub fn clearInterruptDisable(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
|
||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||
@@ -86,6 +148,129 @@ pub const Function = struct {
|
||||
0;
|
||||
return .{ .config = self.config, .cursor = first };
|
||||
}
|
||||
|
||||
/// First capability with `id`, or null.
|
||||
pub fn findCapability(self: *const Function, id: pci_class.CapabilityId) ?Capability {
|
||||
var walk = self.capabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == @intFromEnum(id)) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Program the MSI capability with the kernel's `msi_bind` result and enable it —
|
||||
/// one vector (multiple-message-enable 0, matching the kernel's single-vector
|
||||
/// grant), INTx suppressed. false if the function has no MSI capability.
|
||||
pub fn programMsi(self: *const Function, message: device.Msi) bool {
|
||||
const cap = self.findCapability(.msi) orelse return false;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
const control = mmio.readRegister(u16, control_at);
|
||||
// Program the registers while the capability is disabled.
|
||||
mmio.writeRegister(u16, control_at, control & ~pci_class.msi.control_enable);
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address, @truncate(message.address));
|
||||
const data_offset = if (control & pci_class.msi.control_64bit_capable != 0) offset: {
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address_high, @intCast(message.address >> 32));
|
||||
break :offset pci_class.msi.data_64;
|
||||
} else pci_class.msi.data_32;
|
||||
// Message data is a 16-bit register in both layouts.
|
||||
mmio.writeRegister(u16, cap.offset + data_offset, @truncate(message.data));
|
||||
mmio.writeRegister(u16, control_at, (control & ~pci_class.msi.control_multiple_message_enable_mask) | pci_class.msi.control_enable);
|
||||
self.setInterruptDisable();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Clear the MSI enable bit. No-op if the function has no MSI capability.
|
||||
pub fn disableMsi(self: *const Function) void {
|
||||
const cap = self.findCapability(.msi) orelse return;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) & ~pci_class.msi.control_enable);
|
||||
}
|
||||
|
||||
/// The function's MSI-X capability with its vector table mapped: the table's BIR is
|
||||
/// resolved through `mapBar` (a free cache hit when it is a BAR the driver already
|
||||
/// mapped). null if the capability is absent or the table's BAR cannot be mapped.
|
||||
pub fn msix(self: *Function) ?MsiX {
|
||||
const cap = self.findCapability(.msix) orelse return null;
|
||||
const control = mmio.readRegister(u16, cap.offset + pci_class.msix.control);
|
||||
const word = mmio.readRegister(u32, cap.offset + pci_class.msix.table_offset_word);
|
||||
const location = pci_class.msix.tableLocation(word);
|
||||
const bar_base = self.mapBar(location.bar) orelse return null;
|
||||
return .{
|
||||
.capability = cap.offset,
|
||||
.table = bar_base + location.offset,
|
||||
.entry_count = pci_class.msix.tableSize(control),
|
||||
};
|
||||
}
|
||||
|
||||
/// Bring the function to D0. Firmware can leave a non-boot device in D3hot, where
|
||||
/// its BARs and MSI registers do not decode; call this before touching either. No
|
||||
/// power-management capability means the function is always at D0: nothing to do.
|
||||
/// Preserves PME-Enable and never clears the write-1-to-clear PME-Status bit.
|
||||
pub fn ensurePowerStateD0(self: *const Function) void {
|
||||
const cap = self.findCapability(.power_management) orelse return;
|
||||
const at = cap.offset + pci_class.power_management.control_status;
|
||||
const pmcsr = mmio.readRegister(u16, at);
|
||||
if (pmcsr & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0) return;
|
||||
// PME-Status is RW1C: echoing a read 1 back would clear it, so write it as 0.
|
||||
mmio.writeRegister(u16, at, (pmcsr & ~pci_class.power_management.control_status_power_state_mask & ~pci_class.power_management.control_status_pme_status) | pci_class.power_management.power_state_d0);
|
||||
time.sleepMillis(d0_recovery_millis);
|
||||
}
|
||||
|
||||
/// Function Level Reset via the PCI Express capability: return the hardware to a
|
||||
/// known state (a supervisor re-claiming a device after its driver died, or a driver
|
||||
/// recovering a wedged function). The six BAR dwords are saved and restored — FLR
|
||||
/// clears them, and the bus enumerator's assignment must survive for the descriptor
|
||||
/// correlation and `mapBar` cache to stay valid. Everything else is reset: command
|
||||
/// enables and MSI/MSI-X programming are gone, so the caller re-runs its whole
|
||||
/// bring-up afterwards. false if the function has no PCI Express capability, does
|
||||
/// not advertise FLR (conventional-PCI Advanced Features FLR is a possible
|
||||
/// follow-up), or never became readable again. Blocks for at least 100 ms.
|
||||
pub fn functionLevelReset(self: *const Function) bool {
|
||||
const cap = self.findCapability(.pci_express) orelse return false;
|
||||
const device_capabilities = mmio.readRegister(u32, cap.offset + pci_class.pci_express.device_capabilities);
|
||||
if (device_capabilities & pci_class.pci_express.device_capabilities_flr == 0) return false;
|
||||
|
||||
// Stop new DMA, then give in-flight transactions a bounded chance to drain —
|
||||
// and reset anyway on timeout, since resetting a stuck function is the point.
|
||||
self.disableBusMaster();
|
||||
var waited: u64 = 0;
|
||||
while (mmio.readRegister(u16, cap.offset + pci_class.pci_express.device_status) & pci_class.pci_express.device_status_transactions_pending != 0) {
|
||||
if (waited >= flr_pending_timeout_millis) break;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
|
||||
var bars: [6]u32 = undefined;
|
||||
for (&bars, 0..) |*bar, index| bar.* = mmio.readRegister(u32, self.config + pci_class.config_bar0 + index * 4);
|
||||
|
||||
const control_at = cap.offset + pci_class.pci_express.device_control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) | pci_class.pci_express.device_control_initiate_flr);
|
||||
time.sleepMillis(flr_settle_millis);
|
||||
|
||||
waited = 0;
|
||||
while (self.vendorId() == 0xFFFF) {
|
||||
if (waited >= flr_ready_timeout_millis) return false;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
for (bars, 0..) |bar, index| mmio.writeRegister(u32, self.config + pci_class.config_bar0 + index * 4, bar);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Iterate the extended (PCI Express) capability list at 0x100.. in the 4 KiB ECAM
|
||||
/// page. Empty on a conventional-PCI function (the space reads as all-ones).
|
||||
pub fn extendedCapabilities(self: *const Function) ExtendedCapabilityIterator {
|
||||
return .{ .config = self.config };
|
||||
}
|
||||
|
||||
/// First extended capability with `id`, or null.
|
||||
pub fn findExtendedCapability(self: *const Function, id: u16) ?ExtendedCapability {
|
||||
var walk = self.extendedCapabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == id) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||
@@ -106,3 +291,94 @@ pub const CapabilityIterator = struct {
|
||||
return .{ .id = id, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
/// A resolved MSI-X capability from `Function.msix`: `capability` is the absolute
|
||||
/// virtual address of the config-space header, `table` of vector-table entry 0 (in BAR
|
||||
/// space — table writes are MMIO, not config space). Entries reset masked; bring-up
|
||||
/// order is programEntry per vector, unmaskEntry per used vector, `enable`, then
|
||||
/// `Function.setInterruptDisable`.
|
||||
pub const MsiX = struct {
|
||||
capability: usize,
|
||||
table: usize,
|
||||
entry_count: u16,
|
||||
|
||||
/// Write `message` into table entry `entry`, leaving the entry masked (its reset
|
||||
/// state) — the spec requires masking while address/data change. false if `entry`
|
||||
/// is out of range.
|
||||
pub fn programEntry(self: *const MsiX, entry: u16, message: device.Msi) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size;
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_vector_control, pci_class.msix.entry_vector_control_masked);
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address, @truncate(message.address));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address_high, @intCast(message.address >> 32));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_data, message.data);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the entry's vector-control mask bit — its interrupt is held off (pended in
|
||||
/// the PBA, not lost). false if `entry` is out of range.
|
||||
pub fn maskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, true);
|
||||
}
|
||||
/// Clear the entry's vector-control mask bit. false if `entry` is out of range.
|
||||
pub fn unmaskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, false);
|
||||
}
|
||||
fn writeEntryMask(self: *const MsiX, entry: u16, masked: bool) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size + pci_class.msix.entry_vector_control;
|
||||
const control = mmio.readRegister(u32, at);
|
||||
mmio.writeRegister(u32, at, if (masked)
|
||||
control | pci_class.msix.entry_vector_control_masked
|
||||
else
|
||||
control & ~pci_class.msix.entry_vector_control_masked);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the function-mask control bit: every vector masked regardless of entry bits.
|
||||
pub fn setFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, true);
|
||||
}
|
||||
/// Clear the function-mask control bit.
|
||||
pub fn clearFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, false);
|
||||
}
|
||||
/// Set MSI-X Enable. The caller also calls `Function.setInterruptDisable` (INTx off).
|
||||
pub fn enable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, true);
|
||||
}
|
||||
/// Clear MSI-X Enable.
|
||||
pub fn disable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, false);
|
||||
}
|
||||
fn writeControl(self: *const MsiX, bit: u16, set: bool) void {
|
||||
const at = self.capability + pci_class.msix.control;
|
||||
const control = mmio.readRegister(u16, at);
|
||||
mmio.writeRegister(u16, at, if (set) control | bit else control & ~bit);
|
||||
}
|
||||
};
|
||||
|
||||
/// One extended capability. `offset` is the ABSOLUTE virtual address of its header,
|
||||
/// like `Capability.offset`.
|
||||
pub const ExtendedCapability = struct { id: u16, version: u4, offset: usize };
|
||||
|
||||
pub const ExtendedCapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u16 = @intCast(pci_class.extended_capability_start),
|
||||
guard: u32 = 0, // bounds a malformed chain (480 = the 0xF00-byte space / 8-byte minimum spacing)
|
||||
|
||||
pub fn next(self: *ExtendedCapabilityIterator) ?ExtendedCapability {
|
||||
if (self.cursor == 0 or self.guard >= 480) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const header = pci_class.ExtendedCapabilityHeader.decode(mmio.readRegister(u32, at));
|
||||
// Id 0 marks an empty list; all-ones is a conventional-PCI function (no
|
||||
// extended space — reads come back as FFs).
|
||||
if (header.id == 0 or header.id == 0xFFFF) return null;
|
||||
// A next pointer below 0x100 would walk into the legacy header; treat it as the
|
||||
// terminator it must be (0 is the normal one). The 0xFFC decode mask already
|
||||
// keeps `config + cursor + 4` inside the 4 KiB page.
|
||||
self.cursor = if (header.next >= pci_class.extended_capability_start) header.next else 0;
|
||||
return .{ .id = header.id, .version = header.version, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
@@ -0,0 +1,343 @@
|
||||
//! The device registry: parse `/etc/devices.csv` into match rules and bind a
|
||||
//! reported device to a driver. This is the data-driven replacement for the
|
||||
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||
//! — a device that no row matches goes unbound (logged), never guessed.
|
||||
//!
|
||||
//! Pure logic: no hardware access, no syscalls, no allocator. `parse` fills a
|
||||
//! caller-provided `[]Rule` whose string fields (`hid`, `driver`) are slices
|
||||
//! *into the CSV source*, so the source buffer must outlive the rules (the
|
||||
//! manager holds it in a static buffer for the life of the process — zero-copy).
|
||||
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||
//!
|
||||
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||
//! `/etc/devices.csv` header itself): one rule per line, nine comma-separated
|
||||
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||
//!
|
||||
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
//!
|
||||
//! `bus` is `pci`/`usb`/`acpi`; the numeric fields are hex (with or without a
|
||||
//! `0x` prefix); `*` or an empty field is a wildcard (matches anything). For PCI
|
||||
//! the class triple is base/subclass/prog-IF; for USB it is class/subclass/
|
||||
//! protocol with vendor/device the idVendor/idProduct; ACPI matches on `hid`
|
||||
//! (e.g. "PNP0303") with the triple left blank. `driver` is a full ramdisk path.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
|
||||
/// Which bus a rule or a reported device belongs to. `unknown` is what an
|
||||
/// unrecognised `bus` token parses to — such a rule never matches (its bus
|
||||
/// equals no real device's), so a typo fails safe rather than binding wrongly.
|
||||
pub const Bus = enum {
|
||||
pci,
|
||||
usb,
|
||||
acpi,
|
||||
unknown,
|
||||
|
||||
pub fn fromToken(token: []const u8) Bus {
|
||||
if (std.mem.eql(u8, token, "pci")) return .pci;
|
||||
if (std.mem.eql(u8, token, "usb")) return .usb;
|
||||
if (std.mem.eql(u8, token, "acpi")) return .acpi;
|
||||
return .unknown;
|
||||
}
|
||||
};
|
||||
|
||||
/// A reported device's full identity, as the manager assembles it from a
|
||||
/// `child_added`: the bus-native class triple plus the numeric ids the widened
|
||||
/// ABI now carries, or the ACPI `_HID` string. Fields a given bus does not have
|
||||
/// are zero / empty (a PCI function has no `hid`; an ACPI device has no vendor).
|
||||
pub const Identity = struct {
|
||||
bus: Bus,
|
||||
base: u8 = 0,
|
||||
subclass: u8 = 0,
|
||||
prog_if: u8 = 0,
|
||||
vendor: u16 = 0,
|
||||
device: u16 = 0,
|
||||
subsystem: u32 = 0,
|
||||
hid: []const u8 = "",
|
||||
};
|
||||
|
||||
/// One parsed registry row. A `null` field is a wildcard — it matches any value
|
||||
/// and contributes nothing to specificity. String fields point into the CSV
|
||||
/// source that was parsed (see the module doc).
|
||||
pub const Rule = struct {
|
||||
bus: Bus,
|
||||
base: ?u8 = null,
|
||||
subclass: ?u8 = null,
|
||||
prog_if: ?u8 = null,
|
||||
vendor: ?u16 = null,
|
||||
device: ?u16 = null,
|
||||
subsystem: ?u32 = null,
|
||||
hid: ?[]const u8 = null,
|
||||
driver: []const u8,
|
||||
};
|
||||
|
||||
/// Specificity weights: how much each pinned field counts toward "most specific
|
||||
/// wins". Doubling from the coarsest (`base`) so that each level outweighs *all*
|
||||
/// coarser levels combined (1+2+4+8+16 = 31 < 32) — a rule that pins `device`
|
||||
/// always beats any rule that does not, no matter how many coarse fields the
|
||||
/// latter pins. `hid` and `device` share the top tier (the user's "hid and
|
||||
/// device weigh heaviest"); they never co-occur, since `hid` is ACPI-only and
|
||||
/// `device` is a PCI/USB numeric id.
|
||||
const weight_base: u32 = 1;
|
||||
const weight_subclass: u32 = 2;
|
||||
const weight_prog_if: u32 = 4;
|
||||
const weight_vendor: u32 = 8;
|
||||
const weight_subsystem: u32 = 16;
|
||||
const weight_device: u32 = 32;
|
||||
const weight_hid: u32 = 32;
|
||||
|
||||
/// The outcome of `matchDriver`: the winning rule's driver path, its specificity,
|
||||
/// and whether another rule tied it at that specificity. `ambiguous` is a
|
||||
/// registry authoring error (two equally-specific rules claiming one device); the
|
||||
/// manager logs it loudly and binds the first, so a shadowed rule is visible
|
||||
/// rather than silently dropped.
|
||||
pub const Match = struct {
|
||||
driver: []const u8,
|
||||
specificity: u32,
|
||||
ambiguous: bool,
|
||||
};
|
||||
|
||||
/// Whether `rule` matches `id`: same bus, and every pinned (non-wildcard) field
|
||||
/// equal. `hid` compares as a string; the rest as integers.
|
||||
fn matches(rule: Rule, id: Identity) bool {
|
||||
if (rule.bus != id.bus) return false;
|
||||
if (rule.base) |b| if (b != id.base) return false;
|
||||
if (rule.subclass) |s| if (s != id.subclass) return false;
|
||||
if (rule.prog_if) |p| if (p != id.prog_if) return false;
|
||||
if (rule.vendor) |v| if (v != id.vendor) return false;
|
||||
if (rule.device) |d| if (d != id.device) return false;
|
||||
if (rule.subsystem) |s| if (s != id.subsystem) return false;
|
||||
if (rule.hid) |h| if (!std.mem.eql(u8, h, id.hid)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The specificity score of a rule — the sum of the weights of its pinned fields.
|
||||
fn specificity(rule: Rule) u32 {
|
||||
var score: u32 = 0;
|
||||
if (rule.base != null) score += weight_base;
|
||||
if (rule.subclass != null) score += weight_subclass;
|
||||
if (rule.prog_if != null) score += weight_prog_if;
|
||||
if (rule.vendor != null) score += weight_vendor;
|
||||
if (rule.device != null) score += weight_device;
|
||||
if (rule.subsystem != null) score += weight_subsystem;
|
||||
if (rule.hid != null) score += weight_hid;
|
||||
return score;
|
||||
}
|
||||
|
||||
/// Bind a reported device to a driver: of every rule that matches `id`, return
|
||||
/// the most specific. `null` when nothing matches (the device goes unbound —
|
||||
/// the authoritative registry does not guess). On an exact specificity tie the
|
||||
/// first such rule in file order wins and `ambiguous` is set.
|
||||
pub fn matchDriver(rules: []const Rule, id: Identity) ?Match {
|
||||
var best: ?Match = null;
|
||||
for (rules) |rule| {
|
||||
if (!matches(rule, id)) continue;
|
||||
const score = specificity(rule);
|
||||
if (best) |current| {
|
||||
if (score > current.specificity) {
|
||||
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||
} else if (score == current.specificity) {
|
||||
// Two equally-specific rules claim this device — keep the first,
|
||||
// flag the ambiguity for the manager to log.
|
||||
best.?.ambiguous = true;
|
||||
}
|
||||
} else {
|
||||
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
// --- parsing -----------------------------------------------------------------
|
||||
|
||||
/// What one CSV line parsed to. `malformed` is a non-comment, non-blank line the
|
||||
/// parser could not read (wrong field count, unparsable number, empty driver) —
|
||||
/// the manager counts these and logs, so a broken registry is loud, not silent.
|
||||
const Line = union(enum) {
|
||||
rule: Rule,
|
||||
ignorable, // blank or comment
|
||||
malformed,
|
||||
};
|
||||
|
||||
/// The result of `parse`: how many rules landed in the caller's buffer, and how
|
||||
/// many non-ignorable lines were malformed (for the manager to log). `truncated`
|
||||
/// is set if there were more valid rules than the buffer could hold.
|
||||
pub const ParseResult = struct {
|
||||
count: usize,
|
||||
malformed: usize,
|
||||
truncated: bool,
|
||||
};
|
||||
|
||||
/// Parse one hex field into `T`, honouring `*`/empty as a wildcard (`null`) and
|
||||
/// an optional `0x` prefix. Returns an error only for a genuinely unparsable
|
||||
/// non-wildcard token, so the caller can mark the whole line malformed.
|
||||
fn parseHexField(comptime T: type, field: []const u8) !?T {
|
||||
const token = std.mem.trim(u8, field, " \t");
|
||||
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||
const digits = if (std.mem.startsWith(u8, token, "0x") or std.mem.startsWith(u8, token, "0X"))
|
||||
token[2..]
|
||||
else
|
||||
token;
|
||||
return try std.fmt.parseInt(T, digits, 16);
|
||||
}
|
||||
|
||||
/// Parse a wildcard-or-string field (the `hid` column): `*`/empty → wildcard.
|
||||
fn parseStringField(field: []const u8) ?[]const u8 {
|
||||
const token = std.mem.trim(u8, field, " \t");
|
||||
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||
return token;
|
||||
}
|
||||
|
||||
/// Classify and (if a rule) parse one line. Split out from `parse` so it can be
|
||||
/// unit-tested directly. `line` is the raw line including no newline.
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
|
||||
// Nine comma-separated fields (csv.fields trims each): bus, base, class,
|
||||
// prog_if, vendor, device, subsystem, hid, driver.
|
||||
var cols: [9][]const u8 = undefined;
|
||||
var count: usize = 0;
|
||||
var it = csv.fields(body);
|
||||
while (it.next()) |field| {
|
||||
if (count >= cols.len) return .malformed; // too many columns
|
||||
cols[count] = field;
|
||||
count += 1;
|
||||
}
|
||||
if (count != cols.len) return .malformed; // too few columns
|
||||
|
||||
const bus = Bus.fromToken(cols[0]);
|
||||
if (bus == .unknown) return .malformed;
|
||||
|
||||
const driver = cols[8];
|
||||
if (driver.len == 0) return .malformed;
|
||||
|
||||
return .{ .rule = .{
|
||||
.bus = bus,
|
||||
.base = parseHexField(u8, cols[1]) catch return .malformed,
|
||||
.subclass = parseHexField(u8, cols[2]) catch return .malformed,
|
||||
.prog_if = parseHexField(u8, cols[3]) catch return .malformed,
|
||||
.vendor = parseHexField(u16, cols[4]) catch return .malformed,
|
||||
.device = parseHexField(u16, cols[5]) catch return .malformed,
|
||||
.subsystem = parseHexField(u32, cols[6]) catch return .malformed,
|
||||
.hid = parseStringField(cols[7]),
|
||||
.driver = driver,
|
||||
} };
|
||||
}
|
||||
|
||||
/// Parse a whole `/etc/devices.csv` into `out_rules`. The string fields of the
|
||||
/// returned rules point into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.rule => |rule| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = rule;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// The worked example from the design: a specific virtio-gpu rule (pins vendor +
|
||||
// device) and a generic display rule (class only) both match the virtio card;
|
||||
// the specific one must win. And a plain VGA adapter still falls to the generic
|
||||
// rule. This is the whole point of widening the ABI to carry vendor/device.
|
||||
test "virtio device rule beats the generic display rule" {
|
||||
const text =
|
||||
\\# bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||
\\pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
|
||||
// The virtio-gpu function: display / other, vendor 1AF4 device 1050.
|
||||
const virtio = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x80, .prog_if = 0x00,
|
||||
.vendor = 0x1AF4, .device = 0x1050,
|
||||
}).?;
|
||||
try testing.expect(!virtio.ambiguous);
|
||||
try testing.expectEqualStrings("/system/drivers/virtio-gpu", virtio.driver);
|
||||
|
||||
// A plain VGA adapter (display / VGA) still binds the generic display driver.
|
||||
const vga = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||
.vendor = 0x1234, .device = 0x1111,
|
||||
}).?;
|
||||
try testing.expectEqualStrings("/system/drivers/display", vga.driver);
|
||||
}
|
||||
|
||||
test "no matching row leaves the device unbound" {
|
||||
const text = "pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus\n";
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
|
||||
// An AHCI controller (mass storage / SATA / AHCI) has no row — unbound.
|
||||
const unmatched = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x01, .subclass = 0x06, .prog_if = 0x01,
|
||||
});
|
||||
try testing.expect(unmatched == null);
|
||||
}
|
||||
|
||||
test "acpi rows match on hid" {
|
||||
const text =
|
||||
\\acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
\\acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||
|
||||
const keyboard = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0303" }).?;
|
||||
try testing.expectEqualStrings("/system/drivers/ps2-bus", keyboard.driver);
|
||||
const nothing = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0A03" });
|
||||
try testing.expect(nothing == null);
|
||||
}
|
||||
|
||||
test "equally specific rules flag ambiguity" {
|
||||
const text =
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-a
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-b
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
const hit = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||
}).?;
|
||||
try testing.expect(hit.ambiguous);
|
||||
try testing.expectEqualStrings("/system/drivers/display-a", hit.driver); // first wins
|
||||
}
|
||||
|
||||
test "comments, blanks, and malformed lines" {
|
||||
const text =
|
||||
\\# a header comment
|
||||
\\
|
||||
\\pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus # trailing comment
|
||||
\\pci, ZZ, 03, 30, *, *, *, *, /system/drivers/broken
|
||||
\\pci, 03, 00, 00, *, *, *, *,
|
||||
\\bogus-bus, *, *, *, *, *, *, *, /system/drivers/x
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the xhci row is valid
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad hex, empty driver, bad bus
|
||||
try testing.expectEqualStrings("/system/drivers/usb-xhci-bus", rules[0].driver);
|
||||
try testing.expect(rules[0].hid == null); // trailing comment stripped, hid still wildcard
|
||||
}
|
||||
@@ -99,6 +99,19 @@ pub const Device = struct {
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// Hand the controller a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc, or forwarded from another process) so it binds that buffer into its
|
||||
/// IOMMU domain. Must be called for every buffer whose physical address this device
|
||||
/// will name in a `bulk` transfer, before the transfer. Harmless (and a no-op
|
||||
/// success) when no IOMMU is enforcing. Returns false on failure.
|
||||
pub fn attachDma(self: *Device, handle: ipc.Handle) bool {
|
||||
var request = usb_transfer_protocol.DmaAttachRequest{ .device_token = self.token };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.DmaAttachReply)]u8 = undefined;
|
||||
const result = ipc.callCap(self.bus, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.DmaAttachReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.DmaAttachReply, reply[0..@sizeOf(usb_transfer_protocol.DmaAttachReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
|
||||
@@ -34,6 +34,14 @@ pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
||||
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
||||
/// once the binding holds its own reference — the 32-slot table is otherwise consumed by
|
||||
/// repeated cap-passing.
|
||||
pub fn close(h: Handle) bool {
|
||||
return !failed(sc.systemCall1(.handle_close, h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
|
||||
@@ -12,12 +12,18 @@ const sc = @import("system-call");
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
/// Ask for a capability handle (in `Region.handle`) so the buffer can be delegated to
|
||||
/// another driver and bound into a device's IOMMU domain (`driver.dmaBind`). A driver's
|
||||
/// private rings don't need it; a buffer whose physical address crosses IPC does.
|
||||
pub const shareable: usize = abi.dma_shareable;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, the `physical` address to
|
||||
/// program into the device's registers, and — when `shareable` was requested — a
|
||||
/// capability `handle` naming the region for delegation (null otherwise).
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
handle: ?usize = null,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
@@ -25,21 +31,23 @@ inline fn failed(r: usize) bool {
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
/// `coherent | shareable`). Returns virtual/physical (and a handle when `shareable`), or
|
||||
/// null on failure. Three return values — virtual in rax, physical in rdx, handle in r8
|
||||
/// — so it needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
var r8: usize = undefined; // out: capability handle (abi.no_cap unless shareable)
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
[r8] "={r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
return .{ .virtual = rax, .physical = rdx, .handle = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
|
||||
@@ -38,6 +38,7 @@ pub const DmaRegion = dma.Region;
|
||||
pub const dma_coherent = dma.coherent;
|
||||
pub const dma_write_combining = dma.write_combining;
|
||||
pub const dma_below_4g = dma.below_4g;
|
||||
pub const dma_shareable = dma.shareable;
|
||||
pub const dmaAlloc = dma.alloc;
|
||||
pub const dmaFree = dma.free;
|
||||
|
||||
|
||||
@@ -7,7 +7,9 @@
|
||||
//! buffer**, named by its physical address — the same physical-address handoff
|
||||
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||
//! reply headers do. (Safe while the IOMMU is unenforced; see docs/driver-model.md.)
|
||||
//! reply headers do. Under an enforcing IOMMU the buffer's physical addresses are
|
||||
//! only reachable by the device once the filesystem has `attach`ed the buffer's
|
||||
//! capability (the block server forwards it to the controller); see docs/driver-model.md.
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// geometry() -> { block_size, block_count }
|
||||
@@ -20,6 +22,11 @@ pub const Operation = enum(u32) {
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
/// attach(): the caller's DMA-region capability rides the call's cap slot; the
|
||||
/// block server forwards it to the controller so the buffer's physical addresses
|
||||
/// (named in later read/write) are reachable by the device under an enforcing
|
||||
/// IOMMU. Call once per buffer before using it in a transfer.
|
||||
attach = 4,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
|
||||
@@ -11,6 +11,19 @@
|
||||
/// startup instead of quiet corruption later.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
||||
/// the manager's /etc/devices.csv matcher knows how to read the report's identity
|
||||
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
||||
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
||||
/// the zero default, so an un-upgraded reporter fails to match rather than
|
||||
/// binding to the wrong bus's rule.
|
||||
pub const BusKind = enum(u8) {
|
||||
unknown = 0,
|
||||
pci = 1,
|
||||
usb = 2,
|
||||
acpi = 3,
|
||||
};
|
||||
|
||||
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
||||
pub const Role = enum(u8) {
|
||||
/// Owns a controller and reports the devices behind it (`child_added`).
|
||||
@@ -66,7 +79,9 @@ pub const reply_size = @sizeOf(HelloReply);
|
||||
/// restarted instance rediscovers and re-reports.
|
||||
pub const ChildAdded = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.child_added),
|
||||
reserved0: u8 = 0,
|
||||
/// A `BusKind` value: which bus reported this child, so the manager reads the
|
||||
/// identity in the right namespace and matches against the right `bus` column.
|
||||
bus: u8 = @intFromEnum(BusKind.unknown),
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
/// The reporting driver's own device (the controller) — the child's parent.
|
||||
@@ -80,6 +95,18 @@ pub const ChildAdded = extern struct {
|
||||
/// manager hands a matched driver as its argv assignment — or `no_device`
|
||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||
device_id: u64 = no_device,
|
||||
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
||||
/// concept (ACPI). Carried so the manager's /etc/devices.csv matcher can bind
|
||||
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
||||
vendor: u16 = 0,
|
||||
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
||||
/// level: this is what lets one virtio-gpu (1AF4:1050) be told from any other
|
||||
/// virtio display function without the driver re-confirming after it is spawned.
|
||||
device: u16 = 0,
|
||||
/// The PCI subsystem id, packed `(subsystem_vendor << 16) | subsystem_device`
|
||||
/// (so it reads vendor-first, matching the CSV's `ssvid:ssid`), or 0 when the
|
||||
/// device has no subsystem id (a bridge, or a non-PCI bus).
|
||||
subsystem: u32 = 0,
|
||||
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
||||
/// discovered by firmware string rather than a numeric bus identity. Empty
|
||||
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
||||
|
||||
@@ -45,6 +45,11 @@ pub const Operation = enum(u32) {
|
||||
control = 1,
|
||||
interrupt_subscribe = 2,
|
||||
bulk = 3,
|
||||
/// dma_attach: a class driver hands the controller a DMA-region capability (riding
|
||||
/// the call's cap slot) so the controller binds that buffer into its IOMMU domain
|
||||
/// and may then DMA to the physical addresses inside it. Needed once per buffer the
|
||||
/// class driver will name in a `bulk` transfer (its own, or one forwarded to it).
|
||||
dma_attach = 4,
|
||||
};
|
||||
|
||||
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||
@@ -138,6 +143,19 @@ pub const BulkReply = extern struct {
|
||||
actual_length: u32,
|
||||
};
|
||||
|
||||
/// dma_attach: the region capability rides the call's cap slot; the body only carries
|
||||
/// the device token (scoping) so the controller knows which caller is attaching.
|
||||
pub const DmaAttachRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.dma_attach),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
};
|
||||
|
||||
pub const DmaAttachReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||
pub const InterruptReport = extern struct {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||
|
||||
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/device-driver-development-guide/input.md)
|
||||
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/device-driver-development/input.md)
|
||||
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||
`character` — a keymap — without danos shipping an X11 runtime.
|
||||
|
||||
@@ -76,6 +76,10 @@ pub const SystemCall = enum(u64) {
|
||||
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
||||
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
||||
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
||||
iommu_fault_drain = 50, // iommu_fault_drain() -> count: drain + log pending IOMMU translation faults (a diagnostic; the count of faults seen this call)
|
||||
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
||||
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
||||
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
||||
_,
|
||||
};
|
||||
|
||||
@@ -113,6 +117,7 @@ pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||
pub const dma_shareable: u64 = 8; // return a capability handle (r8) so the region can be delegated + dma_bound
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||
|
||||
@@ -1,59 +0,0 @@
|
||||
//! /system/drivers/display - the generic display engine driver.
|
||||
//! This driver is a non official driver for GPU vendors like Intel, NVIDIA, AMD. It provides basic
|
||||
//! display engine features to the display engine protocol used by the display server, compositor
|
||||
//! and graphical user interface libraries like Zooeee.
|
||||
//!
|
||||
//! This driver is acts like BUS driver, in that it detects the GPU, its capabilities and loads
|
||||
//! sub-drivers for each device detected. Similar The device manager
|
||||
//! finds display adaptor e.g. over the PCI/ACPI, and passes the buck on to this driver to handle.
|
||||
//!
|
||||
//! The display driver provides the low level part of identifying the device and launching the
|
||||
//! generic device driver for a GPU vendor.
|
||||
//!
|
||||
//! It takes over the framebuffer feature that was setup during system boot.
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const device_manager = @import("driver");
|
||||
const logging = @import("logging");
|
||||
const mmio = @import("mmio");
|
||||
const display_protocol = @import("display-protocol");
|
||||
const scanout_protocol = @import("scanout-protocol");
|
||||
var device_id: u64 = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
// Hello the device manager (role: device — we claim one GPU's PCI function
|
||||
// and serve its display engine; we report no children). Best-effort: without a
|
||||
// manager the driver still runs standalone; when present, the manager marks us
|
||||
// up before the hello deadline and restarts us if we die.
|
||||
_ = device_manager.hello(.device, device_id);
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
_ = reply;
|
||||
|
||||
if (message.len < scanout_protocol.request_size) return 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = logging.write("display: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(256, .{
|
||||
.service = .scanout,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
@@ -1,69 +0,0 @@
|
||||
//! /system/drivers/display/intel-integrated - the intel 985 family display engine driver.
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const logging = @import("logging");
|
||||
const mmio = @import("mmio");
|
||||
const display_protocol = @import("display-protocol");
|
||||
const scanout_protocol = @import("scanout-protocol");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
var device_id: u64 = 0;
|
||||
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
_ = reply;
|
||||
|
||||
if (message.len < scanout_protocol.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(scanout_protocol.Request, message[0..scanout_protocol.request_size]);
|
||||
switch (request.operation) {
|
||||
_ => return 0,
|
||||
// TODO:
|
||||
// @intFromEnum(sp.Operation.present) => return scanoutStatus(reply, presentFull()),
|
||||
// @intFromEnum(sp.Operation.get_modes) => {
|
||||
// var response = sp.ModesReply{ .status = 0, .count = offered_modes.len, .modes = undefined };
|
||||
// for (0..sp.max_modes) |i| {
|
||||
// response.modes[i] = if (i < offered_modes.len)
|
||||
// .{ .width = offered_modes[i].width, .height = offered_modes[i].height }
|
||||
// else
|
||||
// .{ .width = 0, .height = 0 };
|
||||
// }
|
||||
// @memcpy(reply[0..sp.modes_reply_size], std.mem.asBytes(&response));
|
||||
// return sp.modes_reply_size;
|
||||
// },
|
||||
// @intFromEnum(sp.Operation.set_mode) => {
|
||||
// const w = request.width;
|
||||
// const h = request.height;
|
||||
// if (w == 0 or h == 0 or w > max_width or h > max_height) return scanoutStatus(reply, false);
|
||||
// current_width = w;
|
||||
// current_height = h;
|
||||
// return scanoutStatus(reply, setScanoutRect());
|
||||
// },
|
||||
else => return 0,
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = logging.write("display/intel-985: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(256, .{
|
||||
.service = .scanout,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||
//! the device-tree id as argv[1]; claims that device and no other.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const memory = @import("memory");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
|
||||
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||
const message_maximum = 256;
|
||||
|
||||
var controller_id: u64 = 0;
|
||||
var register_base: usize = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint; // needed later, for irq binding and timers
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.err("unable to claim device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the device's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||
defer memory.allocator().free(buffer);
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Log every resource BEFORE choosing one (see step 5).
|
||||
var register_index: u64 = 0;
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||
index, resource.kind, resource.start, resource.len,
|
||||
});
|
||||
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||
}
|
||||
if (register_index == 0) {
|
||||
std.log.err("register BAR not found", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
std.log.err("mmio_map failed", .{});
|
||||
return false;
|
||||
};
|
||||
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
std.log.err("missing device id (argv[1])", .{});
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.err("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
// .on_notification only once an IRQ or timer is bound
|
||||
});
|
||||
}
|
||||
@@ -21,19 +21,20 @@ const logging = @import("logging");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
/// Log a discovered function with its (class / subclass / prog-IF) triple decoded
|
||||
/// to human names — the boot-log breadcrumb that says *what* the hardware is, so
|
||||
/// "class 0x01 (Mass Storage Controller) subclass 0x06 (Serial ATA Controller)
|
||||
/// progif 0x01 (AHCI 1.0)" reads straight off the log when writing a new driver.
|
||||
/// A dedicated wider buffer than `writeLine`'s, since the decoded names are long.
|
||||
fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
||||
/// Log a discovered function as its would-be /etc/devices.csv columns (bus, base,
|
||||
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
||||
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
||||
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
||||
/// wildcard. All read unclaimed, through the bridge's ECAM: the enumerator never
|
||||
/// claims the functions it probes (pci.zig's header — the device-owned pci.Function
|
||||
/// view is what needs a claim, not this one). A wide buffer: the names are long.
|
||||
fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32, vendor_id: u16, product_id: u16, subsystem: u32) void {
|
||||
const cc = pci_class.ClassCode.unpack(@truncate(class_triple));
|
||||
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||
var line: [200]u8 = undefined;
|
||||
const text = if (pif.len != 0)
|
||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||
var sub_buffer: [8]u8 = undefined;
|
||||
const sub = if (subsystem == 0) "*" else std.fmt.bufPrint(&sub_buffer, "{X:0>8}", .{subsystem}) catch "*";
|
||||
var line: [320]u8 = undefined;
|
||||
const text = std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} bus=pci base={X:0>2} class={X:0>2} prog_if={X:0>2} vendor={X:0>4} device={X:0>4} subsystem={s} — {s} / {s}{s}{s}\n", .{ bus, dev, function, cc.base, cc.subclass, cc.prog_if, vendor_id, product_id, sub, pci_class.className(cc.base), pci_class.subclassName(cc.base, cc.subclass), if (pif.len != 0) " / " else "", pif }) catch return;
|
||||
_ = logging.write(text);
|
||||
}
|
||||
|
||||
@@ -136,7 +137,6 @@ fn scan() void {
|
||||
if (vendor_device & 0xFFFF == 0xFFFF) continue;
|
||||
const class_revision = configRead(bus, dev, function, 0x08);
|
||||
found += 1;
|
||||
logFunction(bus, dev, function, class_revision >> 8);
|
||||
registerAndReport(bus, dev, function, class_revision >> 8);
|
||||
}
|
||||
}
|
||||
@@ -153,6 +153,12 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||
descriptor.pci_class = class_triple;
|
||||
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
||||
// device. These carry to the manager's /etc/devices.csv matcher so a function
|
||||
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
||||
const vendor_device = configRead(bus, dev, function, 0x00);
|
||||
descriptor.vendor = @truncate(vendor_device);
|
||||
descriptor.device = @truncate(vendor_device >> 16);
|
||||
descriptor.resources[0] = .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = ecam_physical + (((bus - start_bus) << 20) | (dev << 15) | (function << 12)),
|
||||
@@ -164,6 +170,11 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
// write all-ones, read the writable mask back, restore. Header type 0 only.
|
||||
const header_type = (configRead(bus, dev, function, 0x0C) >> 16) & 0x7F;
|
||||
if (header_type == 0) {
|
||||
// Subsystem id lives at 0x2C only on type-0 (device) headers, not on
|
||||
// bridges: dword low half is subsystem-vendor, high half subsystem-device.
|
||||
// Repack vendor-first so it reads like the CSV's `ssvid:ssid`.
|
||||
const subsystem_dword = configRead(bus, dev, function, 0x2C);
|
||||
descriptor.subsystem = (@as(u32, @truncate(subsystem_dword)) << 16) | @as(u32, @truncate(subsystem_dword >> 16));
|
||||
const command = configRead16(bus, dev, function, 0x04);
|
||||
configWrite16(bus, dev, function, 0x04, command & ~@as(u16, 0b11));
|
||||
var i: u64 = 0;
|
||||
@@ -210,15 +221,23 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
configWrite16(bus, dev, function, 0x04, command);
|
||||
}
|
||||
|
||||
// The devices.csv-column + friendly-name breadcrumb, now that vendor/device/
|
||||
// subsystem are read. Every discovered function is logged, matched or not.
|
||||
logFunction(bus, dev, function, class_triple, descriptor.vendor, descriptor.device, descriptor.subsystem);
|
||||
|
||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
||||
return;
|
||||
};
|
||||
const report = device_manager_protocol.ChildAdded{
|
||||
.bus = @intFromEnum(device_manager_protocol.BusKind.pci),
|
||||
.parent = bridge_id,
|
||||
.bus_address = (bus << 8) | (dev << 3) | function,
|
||||
.identity = class_triple,
|
||||
.device_id = registered,
|
||||
.vendor = descriptor.vendor,
|
||||
.device = descriptor.device,
|
||||
.subsystem = descriptor.subsystem,
|
||||
};
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||
|
||||
@@ -90,9 +90,22 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
};
|
||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
// Shareable, so each buffer's capability can be handed to the controller: usb-storage
|
||||
// owns no device, so its buffers are not auto-bound anywhere — the controller reaches
|
||||
// them only once attached. (No-op binding when no IOMMU is enforcing.)
|
||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
for ([_]memory.DmaRegion{ command_wrapper, status_wrapper, command_data }) |region| {
|
||||
if (region.handle) |handle| {
|
||||
if (!device.attachDma(handle)) {
|
||||
_ = logging.write("/system/drivers/usb-storage: could not attach a DMA buffer to the controller\n");
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
}
|
||||
|
||||
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
||||
// with REQUEST SENSE), identify it, and read its capacity.
|
||||
@@ -135,10 +148,17 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// caller's DMA buffer (named by physical address).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < block_protocol.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(block_protocol.Operation.attach) => {
|
||||
// The filesystem's DMA buffer: forward its capability to the controller so
|
||||
// the device can reach it, then release our copy (the binding holds a ref).
|
||||
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
||||
const ok = device.attachDma(handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||
},
|
||||
|
||||
@@ -28,18 +28,37 @@ const usb_ids = @import("usb-ids");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
const library = @import("usb-xhci-library.zig");
|
||||
const pci = @import("pci");
|
||||
|
||||
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||
var controller: ?library.Controller = null;
|
||||
|
||||
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||
/// requests, signals, and the interrupt-poll timer all arrive.
|
||||
/// requests, signals, MSI notifications, and the poll/reconcile timer all arrive.
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
||||
/// re-armed each tick. Frequent enough for responsive input.
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz) when
|
||||
/// polling, re-armed each tick. Frequent enough for responsive input.
|
||||
const poll_interval_ms: u64 = 8;
|
||||
|
||||
/// The timer interval in MSI mode: the ring is drained at interrupt time, and the tick
|
||||
/// only reconciles root ports (real hardware delivers late USB2 companion-hub debounce
|
||||
/// with no reliable port-change event — see onNotification) and un-wedges a lost MSI
|
||||
/// edge (edge-triggered, no kernel mask/ack: a missed IP clear stalls, never storms).
|
||||
const reconcile_interval_ms: u64 = 250;
|
||||
|
||||
/// The controller's own descriptor, kept at file scope because `pci.Function` holds a
|
||||
/// pointer to it for the whole bring-up.
|
||||
var controller_descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
/// Non-null iff MSI mode is active: the vector whose notification badge means "the
|
||||
/// controller interrupted". Null means the 8 ms polling fallback is running.
|
||||
var msi_vector: ?u32 = null;
|
||||
|
||||
fn timerInterval() u64 {
|
||||
return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms;
|
||||
}
|
||||
|
||||
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||
const Open = struct {
|
||||
@@ -95,6 +114,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
controller_descriptor = descriptor;
|
||||
|
||||
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||
@@ -118,6 +138,14 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
};
|
||||
|
||||
// Message-signalled interrupt setup comes BEFORE the controller bring-up, not
|
||||
// after: Controller.init writes IMAN.IE, and QEMU's xhci only registers the MSI-X
|
||||
// vector as in-use when that write happens with MSI-X already enabled (its
|
||||
// intr_update callback early-outs on !msix_enabled, and msix_notify silently
|
||||
// drops interrupts for an unused vector). Real hardware does not care about the
|
||||
// order; QEMU requires it.
|
||||
setupMsi();
|
||||
|
||||
// Bring the controller up: reset it, stand up the command and event rings,
|
||||
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||
controller = library.Controller.init(register_base) orelse {
|
||||
@@ -146,12 +174,52 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
|
||||
scanPorts(handle);
|
||||
|
||||
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
||||
// re-armed on each tick in onNotification; class drivers subscribe later.
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||
// slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification.
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Switch the event ring from timer polling to message-signalled interrupts, if the
|
||||
/// whole path is available: map the function's config space, bind a vector, program
|
||||
/// the MSI capability — or, on an MSI-X-only function (QEMU's qemu-xhci is one: it
|
||||
/// advertises MSI-X and PCIe but no plain MSI), entry 0 of the MSI-X table, which
|
||||
/// takes the same kernel (address, data) pair (xHCI interrupter 0 raises vector 0).
|
||||
/// Any step failing leaves `msi_vector` null and the 8 ms polling path exactly as it
|
||||
/// was. The controller side needs nothing extra — IMAN.IE and USBCMD.INTE are already
|
||||
/// set (see Controller.init: QEMU only writes runtime events with the interrupter
|
||||
/// enabled).
|
||||
///
|
||||
/// After a supervised kill, the kernel drops the vector binding but the device still
|
||||
/// has the interrupt enabled and fires the stale vector; the kernel EOIs it
|
||||
/// harmlessly, and the respawned driver re-runs this with its fresh vector.
|
||||
fn setupMsi() void {
|
||||
var function = pci.Function.map(controller_id, &controller_descriptor) orelse {
|
||||
std.log.info("config-space map failed; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
function.enableMemoryAndBusMaster();
|
||||
const message = device.msiBind(controller_id, service_endpoint) orelse {
|
||||
std.log.info("msi_bind unavailable; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
if (function.programMsi(message)) {
|
||||
msi_vector = message.data;
|
||||
std.log.info("msi active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
if (function.msix()) |table| {
|
||||
if (table.programEntry(0, message) and table.unmaskEntry(0)) {
|
||||
table.enable();
|
||||
function.setInterruptDisable();
|
||||
msi_vector = message.data;
|
||||
std.log.info("msix active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
}
|
||||
std.log.info("no msi/msi-x capability; polling at {d} ms", .{poll_interval_ms});
|
||||
}
|
||||
|
||||
var register_base: usize = 0;
|
||||
|
||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||
@@ -379,6 +447,7 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
||||
};
|
||||
|
||||
const report = device_manager_protocol.ChildAdded{
|
||||
.bus = @intFromEnum(device_manager_protocol.BusKind.usb),
|
||||
.parent = controller_id,
|
||||
.bus_address = (@as(u64, port) << 8) | interface.number,
|
||||
.identity = identity,
|
||||
@@ -389,13 +458,16 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
||||
std.log.info("child report for port {d} interface {d} failed", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
std.log.info("port {d} interface {d}: {s} ({d}/{d}/{d}) registered as device {d}", .{
|
||||
// The devices.csv columns (bus=usb, and the class triple as base/class/prog_if)
|
||||
// then the human-readable interface name — a would-be /etc/devices.csv row read
|
||||
// straight off the boot log.
|
||||
std.log.info("port {d} interface {d} bus=usb base={X:0>2} class={X:0>2} prog_if={X:0>2} — {s} registered as device {d}", .{
|
||||
port,
|
||||
interface.number,
|
||||
usb_ids.interfaceName(interface.class, interface.subclass, interface.protocol),
|
||||
interface.class,
|
||||
interface.subclass,
|
||||
interface.protocol,
|
||||
usb_ids.interfaceName(interface.class, interface.subclass, interface.protocol),
|
||||
registered,
|
||||
});
|
||||
return registered;
|
||||
@@ -412,10 +484,22 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
||||
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, capability),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
|
||||
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
|
||||
/// binding holds its own kernel reference, so the forwarded capability is closed here.
|
||||
fn handleDmaAttach(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
|
||||
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||
const handle = capability orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||
const ok = device.dmaBind(controller_id, handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
|
||||
}
|
||||
|
||||
fn writeReply(reply: []u8, value: anytype) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
@@ -497,10 +581,27 @@ fn handleBulk(message: []const u8, reply: []u8) usize {
|
||||
return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||
}
|
||||
|
||||
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
||||
/// each to the class driver that subscribed, then re-arm the timer.
|
||||
/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out.
|
||||
/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI);
|
||||
/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event
|
||||
/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being
|
||||
/// swallowed until the reconcile tick.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & ipc.notify_timer_bit == 0) return;
|
||||
if (badge & ipc.notify_timer_bit != 0) {
|
||||
serviceController();
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return;
|
||||
}
|
||||
const vector = msi_vector orelse return;
|
||||
if (badge & ~ipc.notify_badge_bit != vector) return;
|
||||
if (controller) |*engine| engine.acknowledgeInterrupt();
|
||||
serviceController();
|
||||
}
|
||||
|
||||
/// Everything one servicing pass does, shared verbatim by the poll/reconcile tick and
|
||||
/// the MSI notification: drain the event ring, reconcile root ports, service hub
|
||||
/// changes, and push interrupt reports to their class drivers.
|
||||
fn serviceController() void {
|
||||
if (controller) |*engine| {
|
||||
engine.pump();
|
||||
// Poll every root port and reconcile — a device present but not yet
|
||||
@@ -560,7 +661,6 @@ fn onNotification(badge: u64) void {
|
||||
_ = ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||
}
|
||||
}
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
|
||||
@@ -1563,6 +1563,15 @@ pub const Controller = struct {
|
||||
return report;
|
||||
}
|
||||
|
||||
/// Clear interrupter 0's pending bit (IMAN.IP). IP is write-1-to-clear, and the
|
||||
/// read-back carries IE (plain read-write) through unchanged. In MSI mode the
|
||||
/// driver clears IP **before** draining the ring: an event that lands after the
|
||||
/// drain then takes IP 0→1 and fires a fresh edge, where clearing afterwards would
|
||||
/// leave a race in which a new event finds IP already set and raises nothing.
|
||||
pub fn acknowledgeInterrupt(self: *const Controller) void {
|
||||
write32(self.interrupter(interrupter_management), read32(self.interrupter(interrupter_management)) | 1);
|
||||
}
|
||||
|
||||
/// Drain any events currently on the event ring: interrupt reports into the
|
||||
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
||||
/// these were silently dropped before M20). Non-blocking — called on the
|
||||
|
||||
@@ -33,11 +33,6 @@ const vg = @import("virtio-gpu-protocol.zig");
|
||||
/// the compositor in the announce so it packs colours in the surface's byte order.
|
||||
const display_format_bgrx: u32 = 1;
|
||||
|
||||
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
|
||||
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
|
||||
const virtio_vendor: u16 = 0x1AF4;
|
||||
const virtio_gpu_device: u16 = 0x1050;
|
||||
|
||||
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
||||
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
||||
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
||||
@@ -210,19 +205,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
};
|
||||
|
||||
// Config space is resource 0. Confirm it really is a virtio-gpu, then enable memory-space
|
||||
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
||||
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
||||
// Config space is resource 0. The registry (/etc/devices.csv) bound this driver by the
|
||||
// exact virtio-gpu identity (vendor 0x1AF4 / device 0x1050), so there is no re-confirm to
|
||||
// do here any more — map config space and enable memory-space decode + bus mastering (the
|
||||
// device DMAs the ring and backing out of RAM; pci-bus only preserves whatever the firmware
|
||||
// left, and a secondary display is often left disabled).
|
||||
var function = pci.Function.map(device_id, descriptor) orelse {
|
||||
std.log.info("config-space map failed", .{});
|
||||
return false;
|
||||
};
|
||||
const vendor = function.vendorId();
|
||||
const dev = function.deviceId();
|
||||
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
||||
std.log.info("not a virtio-gpu (vendor 0x{x} device 0x{x})", .{ vendor, dev });
|
||||
return false;
|
||||
}
|
||||
function.enableMemoryAndBusMaster();
|
||||
|
||||
// Walk the capability list for the virtio common-config and notify structures (V3 needs
|
||||
@@ -332,6 +323,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("could not resolve the scanout surface physical address", .{});
|
||||
return false;
|
||||
};
|
||||
// Bind the shared surface into this device's IOMMU domain so the GPU may DMA the
|
||||
// framebuffer (attach_backing points it here). The ring/command buffers are
|
||||
// dma_alloc'd and auto-bound; a shared-memory surface needs an explicit bind. We keep
|
||||
// the handle (it is also passed to the display service), so do not close it. No-op
|
||||
// without an IOMMU.
|
||||
if (!device.dmaBind(device_id, surface.handle)) {
|
||||
std.log.info("could not bind the scanout surface for DMA", .{});
|
||||
return false;
|
||||
}
|
||||
{
|
||||
const request = requestAt(vg.ResourceAttachBacking);
|
||||
request.* = .{
|
||||
|
||||
+151
-18
@@ -95,18 +95,52 @@ pub const PlatformInformation = struct {
|
||||
override_count: usize = 0,
|
||||
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
||||
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
|
||||
/// Detection is the first step; per-device domain enforcement lands with the first
|
||||
/// DMA driver.
|
||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16),
|
||||
/// and the kernel says so at every boot (the fail-open platform log line). When
|
||||
/// true, the IOMMU core builds per-device translation domains from this record.
|
||||
iommu_present: bool = false,
|
||||
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
|
||||
/// true when the present unit is AMD-Vi (from IVRS) rather than Intel VT-d (DMAR).
|
||||
/// The two are mutually exclusive on real hardware; the IOMMU core picks the backend.
|
||||
iommu_is_amd: bool = false,
|
||||
/// MMIO base of the selected DMA-remapping hardware unit — the VT-d DRHD with
|
||||
/// INCLUDE_PCI_ALL (the catch-all unit; falls back to the first), or the AMD-Vi
|
||||
/// IOMMU's control-register base from the first IVHD.
|
||||
iommu_base: u64 = 0,
|
||||
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
|
||||
/// reading it back nonzero confirms a real, mappable VT-d unit.
|
||||
iommu_version: u32 = 0,
|
||||
/// The unit's Capability register (offset 0x08): supported address widths, number
|
||||
/// of domains, etc. Recorded now; consumed when enforcement is built.
|
||||
/// of domains, etc. Consumed by the IOMMU core when it enables translation.
|
||||
iommu_capabilities: u64 = 0,
|
||||
/// Whether the selected unit carries INCLUDE_PCI_ALL. False means every unit is
|
||||
/// device-scoped (unusual) — the core still enables on the selected unit but
|
||||
/// devices outside its scope remain untranslated.
|
||||
iommu_include_all: bool = false,
|
||||
/// DRHD units in the DMAR beyond the selected one. Devices scoped to those units
|
||||
/// (typically the integrated GPU) are NOT translated by v1 — the boot log warns.
|
||||
iommu_extra_units: u8 = 0,
|
||||
/// Reserved-memory regions (DMAR RMRRs): firmware-owned buffers a named device
|
||||
/// keeps DMAing into across the OS handoff (classically the xHC keyboard-emulation
|
||||
/// buffer). These must be identity-mapped in the device's domain BEFORE translation
|
||||
/// enables, or platform firmware breaks. Only single-path endpoint scopes are
|
||||
/// recorded; anything fancier is skipped with a loud log at parse time.
|
||||
rmrr: [maximum_rmrr]RmrrRegion = undefined,
|
||||
rmrr_count: usize = 0,
|
||||
/// RMRR device scopes the parser could not record (multi-hop paths, sub-hierarchy
|
||||
/// types, or table overflow). Non-zero means a device keeps an unmapped firmware
|
||||
/// buffer — the kernel boot log warns loudly (the platform module itself is
|
||||
/// log-free by design; it records, the kernel reports).
|
||||
rmrr_skipped: u8 = 0,
|
||||
};
|
||||
|
||||
pub const maximum_rmrr = 8;
|
||||
|
||||
/// One recorded RMRR: the device (requester id) and the inclusive physical range it
|
||||
/// must always be allowed to reach.
|
||||
pub const RmrrRegion = struct {
|
||||
bdf: u16,
|
||||
base: u64,
|
||||
limit: u64,
|
||||
};
|
||||
|
||||
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||
@@ -248,6 +282,8 @@ const SLIT: [4]u8 = "SLIT".*;
|
||||
const SRAT: [4]u8 = "SRAT".*;
|
||||
/// Secondary System Description Table (SSDT)
|
||||
const DMAR: [4]u8 = "DMAR".*;
|
||||
/// I/O Virtualization Reporting Structure (IVRS) — the AMD-Vi analogue of DMAR.
|
||||
const IVRS: [4]u8 = "IVRS".*;
|
||||
const SSDT: [4]u8 = "SSDT".*;
|
||||
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
||||
const SPCR: [4]u8 = "SPCR".*;
|
||||
@@ -467,6 +503,8 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
parseSpcr(header);
|
||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||
parseDmar(hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &IVRS)) {
|
||||
parseIvrs(header);
|
||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||
addAmlBlock(sdt_physical);
|
||||
@@ -757,18 +795,33 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
||||
|
||||
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
||||
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
|
||||
// register base sits at offset 8 within it.
|
||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition): flags byte at
|
||||
// offset 4 (bit 0 = INCLUDE_PCI_ALL, the catch-all unit), 64-bit register base at
|
||||
// offset 8. Type 1 is an RMRR (Reserved Memory Region Reporting): a physical range at
|
||||
// offsets 8/16 (base / inclusive limit) that the device(s) named by the trailing
|
||||
// device-scope entries keep DMAing into across the firmware→OS handoff.
|
||||
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
||||
const dmar_type_drhd: u16 = 0;
|
||||
const dmar_type_rmrr: u16 = 1;
|
||||
const drhd_flags_offset = 4;
|
||||
const drhd_include_pci_all: u8 = 1;
|
||||
const drhd_register_base_offset = 8;
|
||||
const rmrr_base_offset = 8;
|
||||
const rmrr_limit_offset = 16;
|
||||
const rmrr_scopes_offset = 24;
|
||||
// Device-scope entry (within DRHD/RMRR structures): type 1 = PCI endpoint; the path is
|
||||
// (device, function) byte pairs from offset 6, one pair per bridge hop plus the leaf.
|
||||
const scope_type_pci_endpoint: u8 = 1;
|
||||
const scope_start_bus_offset = 5;
|
||||
const scope_path_offset = 6;
|
||||
|
||||
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
|
||||
/// register block, and record its version and capabilities. This is *detection only*:
|
||||
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
|
||||
/// be gated by a per-device translation domain), but no domains are programmed yet —
|
||||
/// enforcement is built with the first DMA driver, which is what there is to protect and
|
||||
/// test against. See docs/driver-model.md (M16), the honest caveat.
|
||||
/// DMAR -> the VT-d unit(s) and reserved memory regions. Walks every remapping
|
||||
/// structure: selects the INCLUDE_PCI_ALL DRHD (the catch-all covering all devices not
|
||||
/// scoped elsewhere — commonly the SECOND unit on real machines, after an iGPU-scoped
|
||||
/// one), counts the rest so the boot log can warn that their devices stay untranslated,
|
||||
/// and records single-path endpoint RMRRs for the IOMMU core to pre-map before it
|
||||
/// enables translation. Multi-hop RMRR scopes are skipped loudly: better a named gap
|
||||
/// than a silent one.
|
||||
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const total: usize = header.length;
|
||||
@@ -780,13 +833,93 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
||||
if (kind == dmar_type_drhd) {
|
||||
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
||||
const include_all = ((fadt(u8, base, total, off + drhd_flags_offset) orelse 0) & drhd_include_pci_all) != 0;
|
||||
if (register_base != 0) {
|
||||
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
||||
// Selection: the INCLUDE_PCI_ALL unit wins; otherwise keep the first
|
||||
// seen. A later catch-all replaces an earlier scoped unit.
|
||||
const replace = !platform_information.iommu_present or
|
||||
(include_all and !platform_information.iommu_include_all);
|
||||
if (replace) {
|
||||
if (platform_information.iommu_present) platform_information.iommu_extra_units += 1;
|
||||
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
||||
platform_information.iommu_present = true;
|
||||
platform_information.iommu_base = register_base;
|
||||
platform_information.iommu_include_all = include_all;
|
||||
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
||||
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
||||
} else {
|
||||
platform_information.iommu_extra_units += 1;
|
||||
}
|
||||
}
|
||||
} else if (kind == dmar_type_rmrr) {
|
||||
parseRmrr(base, total, off, length);
|
||||
}
|
||||
off += length;
|
||||
}
|
||||
}
|
||||
|
||||
/// One RMRR structure: record a {bdf, base, limit} per single-path endpoint scope.
|
||||
fn parseRmrr(base: [*]align(1) const u8, total: usize, off: usize, length: usize) void {
|
||||
const range_base = fadt(u64, base, total, off + rmrr_base_offset) orelse return;
|
||||
const range_limit = fadt(u64, base, total, off + rmrr_limit_offset) orelse return;
|
||||
if (range_limit < range_base) return;
|
||||
|
||||
var scope = off + rmrr_scopes_offset;
|
||||
const end = off + length;
|
||||
while (scope + 6 <= end) {
|
||||
const scope_type = fadt(u8, base, total, scope) orelse break;
|
||||
const scope_length = fadt(u8, base, total, scope + 1) orelse break;
|
||||
if (scope_length < 6 or scope + scope_length > end) break;
|
||||
if (scope_type == scope_type_pci_endpoint and scope_length == scope_path_offset + 2) {
|
||||
// Single (device, function) pair: a directly-reachable endpoint.
|
||||
const bus = fadt(u8, base, total, scope + scope_start_bus_offset) orelse 0;
|
||||
const device = fadt(u8, base, total, scope + scope_path_offset) orelse 0;
|
||||
const function = fadt(u8, base, total, scope + scope_path_offset + 1) orelse 0;
|
||||
if (platform_information.rmrr_count < maximum_rmrr) {
|
||||
platform_information.rmrr[platform_information.rmrr_count] = .{
|
||||
.bdf = (@as(u16, bus) << 8) | (@as(u16, device) << 3) | function,
|
||||
.base = range_base,
|
||||
.limit = range_limit,
|
||||
};
|
||||
platform_information.rmrr_count += 1;
|
||||
} else {
|
||||
platform_information.rmrr_skipped +|= 1; // table full
|
||||
}
|
||||
} else {
|
||||
platform_information.rmrr_skipped +|= 1; // multi-hop path or non-endpoint scope
|
||||
}
|
||||
scope += scope_length;
|
||||
}
|
||||
}
|
||||
|
||||
// IVRS layout (AMD I/O Virtualization spec): 36-byte ACPI header, IVinfo u32 @36,
|
||||
// 8 reserved @40, then IVHD/IVMD blocks from @48. An IVHD common header is type u8 @0,
|
||||
// flags u8 @1, length u16 @2, device id u16 @4, capability offset u16 @6, IOMMU base
|
||||
// address u64 @8, PCI segment u16 @16, IOMMU info u16 @18.
|
||||
const ivrs_blocks_offset = 48;
|
||||
const ivhd_type_10: u8 = 0x10;
|
||||
const ivhd_type_11: u8 = 0x11;
|
||||
const ivhd_base_offset = 8;
|
||||
|
||||
/// IVRS -> detect an AMD-Vi IOMMU. Record the control-register base from the first IVHD
|
||||
/// of type 0x10/0x11. Per-device entries and IVMD (the AMD analogue of RMRR) are ignored
|
||||
/// in v1 — the default-deny device table is what we build anyway, and QEMU emits no IVMD;
|
||||
/// a real machine that needs them is flagged untested on AMD regardless.
|
||||
fn parseIvrs(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const total: usize = header.length;
|
||||
var off: usize = ivrs_blocks_offset;
|
||||
while (off + 4 <= total) {
|
||||
const kind = fadt(u8, base, total, off) orelse break;
|
||||
const length = fadt(u16, base, total, off + 2) orelse break;
|
||||
if (length < 4 or off + length > total) break;
|
||||
if (kind == ivhd_type_10 or kind == ivhd_type_11) {
|
||||
const iommu_base = fadt(u64, base, total, off + ivhd_base_offset) orelse 0;
|
||||
if (iommu_base != 0) {
|
||||
platform_information.iommu_present = true;
|
||||
platform_information.iommu_base = register_base;
|
||||
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
||||
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
||||
return; // first unit is enough for detection; multi-unit is future
|
||||
platform_information.iommu_is_amd = true;
|
||||
platform_information.iommu_base = iommu_base;
|
||||
return; // first IVHD is enough; multi-unit is future work
|
||||
}
|
||||
}
|
||||
off += length;
|
||||
|
||||
@@ -167,6 +167,20 @@ pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the claim on `id` iff `owner` holds it — the rollback for a claim that
|
||||
/// cannot be confined (the IOMMU domain could not be created/attached). Returns true
|
||||
/// when a claim was actually cleared.
|
||||
pub fn unclaim(id: u64, owner: u32) bool {
|
||||
if (id >= count) return false;
|
||||
if (claimed[@intCast(id)]) |o| {
|
||||
if (o == owner) {
|
||||
claimed[@intCast(id)] = null;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Resource `index` of device `id`, or null if out of range.
|
||||
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
if (id >= count) return null;
|
||||
@@ -175,6 +189,50 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
return d.resources[@intCast(index)];
|
||||
}
|
||||
|
||||
/// The PCI requester id (bus<<8 | device<<3 | function) of device `id`, derived from
|
||||
/// its config-space slice against its host bridge's ECAM window — the identity a VT-d
|
||||
/// context entry / AMD-Vi DTE is keyed by. null when `id` is not a PCI function or the
|
||||
/// geometry doesn't decode. The kernel never stored the BDF (the descriptor has no such
|
||||
/// field); pci-bus encodes it into resource 0's physical base as
|
||||
/// `ecam_base + ((bus - start_bus) << 20 | device << 15 | function << 12)`, and the
|
||||
/// requester id the device emits uses the absolute bus, so we add `start_bus << 8` back.
|
||||
pub fn pciAddressOf(id: u64) ?u16 {
|
||||
if (id >= count) return null;
|
||||
const d = &devices[@intCast(id)];
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) return null;
|
||||
if (d.resource_count == 0) return null;
|
||||
const config = d.resources[0];
|
||||
if (config.kind != @intFromEnum(device_abi.ResourceKind.memory) or config.len != 4096) return null;
|
||||
|
||||
// Walk up to the host bridge, whose resource 0 is the segment's ECAM window and
|
||||
// resource 1 the bus_range (start_bus, bus_count).
|
||||
var parent = d.parent;
|
||||
while (parent != device_abi.no_parent and parent < count) {
|
||||
const p = &devices[@intCast(parent)];
|
||||
if (p.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) {
|
||||
if (p.resource_count < 2) return null;
|
||||
const ecam = p.resources[0];
|
||||
const bus_range = p.resources[1];
|
||||
if (config.start < ecam.start or config.start >= ecam.start + ecam.len) return null;
|
||||
const offset = config.start - ecam.start;
|
||||
const start_bus: u16 = @intCast(bus_range.start & 0xFF);
|
||||
return @intCast((offset >> 12) + (@as(u64, start_bus) << 8));
|
||||
}
|
||||
parent = p.parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Call `visit(id, bdf)` for every PCI function in the table — the IOMMU core's boot
|
||||
/// sweep to place every device under a domain. Only functions whose BDF decodes are
|
||||
/// visited.
|
||||
pub fn forEachPciFunction(visit: *const fn (id: u64, bdf: u16) void) void {
|
||||
var id: u64 = 0;
|
||||
while (id < count) : (id += 1) {
|
||||
if (pciAddressOf(id)) |bdf| visit(id, bdf);
|
||||
}
|
||||
}
|
||||
|
||||
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
//! system/kernel/iommu-amd.zig — AMD-Vi (AMD I/O Virtualization) backend for the IOMMU
|
||||
//! core. The AMD analogue of iommu-intel.zig: it supplies the core's `Backend` vtable
|
||||
//! with AMD-Vi's page-table entry bits and drives the device table, command buffer, and
|
||||
//! event log.
|
||||
//!
|
||||
//! **UNTESTED on real AMD hardware.** danos is developed on Intel; this backend is
|
||||
//! validated only against QEMU's `-device amd-iommu,dma-remap=on`, whose AMD-Vi
|
||||
//! emulation is far less exercised than its Intel one. Every code path here should be
|
||||
//! read as "QEMU-verified, real-AMD-unverified" until an AMD machine confirms it.
|
||||
//!
|
||||
//! Interrupt remapping is left off (the DTE forwards interrupts unmapped), so MSI writes
|
||||
//! to the 0xFEE00000 range reach the APIC untranslated, exactly as on the Intel path.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const architecture = @import("architecture");
|
||||
const log = @import("log.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// MMIO register offsets from the IOMMU control-register base.
|
||||
const reg_device_table_base = 0x00; // base | (pages-1) in bits 8:0
|
||||
const reg_command_buffer_base = 0x08; // base | (ComLen << 56)
|
||||
const reg_event_log_base = 0x10; // base | (EventLen << 56)
|
||||
const reg_control = 0x18;
|
||||
const reg_command_head = 0x2000;
|
||||
const reg_command_tail = 0x2008;
|
||||
const reg_event_head = 0x2010;
|
||||
const reg_event_tail = 0x2018;
|
||||
const reg_status = 0x2020;
|
||||
|
||||
const control_iommu_enable: u64 = 1 << 0;
|
||||
const control_event_log_enable: u64 = 1 << 2;
|
||||
const control_command_buffer_enable: u64 = 1 << 12;
|
||||
|
||||
// Device table: one 32-byte DTE (4 qwords) per requester id, indexed by bdf.
|
||||
const device_table_pages = 512; // 2 MiB = 65536 entries (a full 256-bus segment)
|
||||
const dte_qwords = 4;
|
||||
const dte_valid: u64 = 1 << 0; // V
|
||||
const dte_translation_valid: u64 = 1 << 1; // TV
|
||||
const dte_mode_shift = 9; // bits 11:9 — page-table levels
|
||||
const dte_read: u64 = 1 << 61; // IR
|
||||
const dte_write: u64 = 1 << 62; // IW
|
||||
const dte_intctl_forward: u64 = @as(u64, 1) << 60; // qword2 bits 61:60 = 01b: forward interrupts unmapped
|
||||
|
||||
// Page-table entry bits (AMD native format).
|
||||
const pte_present: u64 = 1 << 0; // PR
|
||||
const pte_next_level_shift = 9; // bits 11:9: 0 = leaf, N = pointer to a level-N table
|
||||
const pte_read: u64 = 1 << 61; // IR
|
||||
const pte_write: u64 = 1 << 62; // IW
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// Command buffer / event log: one 4 KiB frame each = 256 entries.
|
||||
const ring_entries = 256;
|
||||
const ring_length_code: u64 = 8; // log2(256), the ComLen/EventLen field value
|
||||
const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
||||
const command_completion_wait: u64 = 0x01;
|
||||
const command_invalidate_devtab: u64 = 0x02;
|
||||
const command_invalidate_pages: u64 = 0x03;
|
||||
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
||||
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
||||
|
||||
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
||||
|
||||
var register_base: usize = 0;
|
||||
var device_table: u64 = 0; // physical base of the device table
|
||||
var command_buffer: u64 = 0;
|
||||
var event_log: u64 = 0;
|
||||
var completion_frame: u64 = 0; // COMPLETION_WAIT store target
|
||||
var command_tail: u32 = 0; // our software copy of the command tail (bytes)
|
||||
|
||||
var fault_log_budget: u32 = 32;
|
||||
var completion_warned = false;
|
||||
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn ram(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, allocate the device table / command buffer / event log.
|
||||
/// Returns the vtable, or null if the boot-time allocations fail.
|
||||
pub fn detect(info: platform.PlatformInformation) ?iommu.Backend {
|
||||
register_base = architecture.mapMmio(info.iommu_base, 16 * 1024, true);
|
||||
|
||||
device_table = pmm.allocContiguous(device_table_pages, ~@as(u64, 0)) orelse return null;
|
||||
zero(device_table, device_table_pages); // all-zero DTE = V=0 = deny every device
|
||||
command_buffer = allocZeroedFrame() orelse return null;
|
||||
event_log = allocZeroedFrame() orelse return null;
|
||||
completion_frame = allocZeroedFrame() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = false, // 4 KiB leaves only (AMD superpage encoding deferred)
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the base registers and enable translation. The device table is already
|
||||
/// zeroed (every device denied) except any entries `attach` wrote for RMRR/claimed
|
||||
/// devices, so turning translation on blocks all other DMA and logs it.
|
||||
pub fn enable() void {
|
||||
write64(reg_device_table_base, (device_table & address_mask) | (device_table_pages - 1));
|
||||
write64(reg_command_buffer_base, (command_buffer & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_event_log_base, (event_log & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_command_head, 0);
|
||||
write64(reg_command_tail, 0);
|
||||
write64(reg_event_head, 0);
|
||||
write64(reg_event_tail, 0);
|
||||
command_tail = 0;
|
||||
// Buffers first, then the master enable.
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable);
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable | control_iommu_enable);
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
_ = huge; // 4 KiB only
|
||||
return (physical & address_mask) | pte_present | pte_read | pte_write; // Next Level 0 = leaf
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
// This entry (at `level`) points to a table one level down; AMD's Next Level names
|
||||
// the pointed-to table's level.
|
||||
const next_level: u64 = @as(u64, level) - 1;
|
||||
return (table_physical & address_mask) | pte_present | pte_read | pte_write | (next_level << pte_next_level_shift);
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & pte_present) != 0;
|
||||
}
|
||||
fn flushStructure(address: usize) void {
|
||||
_ = address; // AMD-Vi reads its structures coherently
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = (page_table_root & address_mask) | dte_valid | dte_translation_valid |
|
||||
(@as(u64, levels) << dte_mode_shift) | dte_read | dte_write;
|
||||
dte[1] = @as(u64, domain); // DomainID in bits 15:0
|
||||
dte[2] = dte_intctl_forward; // forward interrupts unmapped (no remapping)
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = 0; // V=0: deny
|
||||
dte[1] = 0;
|
||||
dte[2] = 0;
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
submitCommand((command_invalidate_pages << command_opcode_shift) | (@as(u64, domain) << 32), invalidate_pages_all);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const tail = read64(reg_event_tail) & 0xFFFF_FFF0;
|
||||
var head = read64(reg_event_head) & 0xFFFF_FFF0;
|
||||
if (head == tail) return 0;
|
||||
var seen: usize = 0;
|
||||
while (head != tail) {
|
||||
const entry = ram(event_log) + (head / 8);
|
||||
const code: u4 = @truncate(entry[0] >> command_opcode_shift);
|
||||
if (code == 0x2) { // IO_PAGE_FAULT
|
||||
const source: u16 = @truncate(entry[0]);
|
||||
logFault(source, entry[1]);
|
||||
}
|
||||
seen += 1;
|
||||
head += 16;
|
||||
if (head >= ring_entries * 16) head = 0;
|
||||
}
|
||||
write64(reg_event_head, head);
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64) void {
|
||||
if (fault_log_budget == 0) return;
|
||||
fault_log_budget -= 1;
|
||||
log.print("DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=amd-io-page-fault\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
});
|
||||
if (fault_log_budget == 0) log.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
}
|
||||
|
||||
// --- command ring --------------------------------------------------------------------
|
||||
|
||||
fn invalidateDevice(bdf: u16) void {
|
||||
submitCommand((command_invalidate_devtab << command_opcode_shift) | @as(u64, bdf), 0);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
/// Append a 128-bit command (two qwords) to the ring and advance the tail.
|
||||
fn submitCommand(qword0: u64, qword1: u64) void {
|
||||
const slot = ram(command_buffer) + (command_tail / 8);
|
||||
slot[0] = qword0;
|
||||
slot[1] = qword1;
|
||||
command_tail += 16;
|
||||
if (command_tail >= ring_entries * 16) command_tail = 0;
|
||||
write64(reg_command_tail, command_tail);
|
||||
}
|
||||
|
||||
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
||||
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
||||
/// register write, so by the time we poll the prior invalidation is already applied; the
|
||||
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
||||
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
||||
/// invalidation itself has happened.
|
||||
fn completeAndWait() void {
|
||||
const sentinel: u64 = 0xC0FFEE;
|
||||
ram(completion_frame)[0] = 0;
|
||||
submitCommand(
|
||||
(command_completion_wait << command_opcode_shift) | (completion_frame & 0x000F_FFFF_FFFF_FFF8) | completion_wait_store,
|
||||
sentinel,
|
||||
);
|
||||
var spins: u64 = 0;
|
||||
while (@as(*const volatile u64, @ptrFromInt(boot_handoff.physicalToVirtual(completion_frame))).* != sentinel) {
|
||||
spins += 1;
|
||||
if (spins > 100_000) {
|
||||
if (!completion_warned) {
|
||||
completion_warned = true;
|
||||
log.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn allocZeroedFrame() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
zero(frame, 1);
|
||||
return frame;
|
||||
}
|
||||
fn zero(physical: u64, pages: usize) void {
|
||||
const words = ram(physical);
|
||||
var i: usize = 0;
|
||||
while (i < pages * page_size / 8) : (i += 1) words[i] = 0;
|
||||
}
|
||||
@@ -0,0 +1,314 @@
|
||||
//! system/kernel/iommu-intel.zig — Intel VT-d backend for the IOMMU core.
|
||||
//!
|
||||
//! Provides the core (iommu.zig) with the VT-d hardware specifics behind its `Backend`
|
||||
//! vtable: second-level page-table entry bits, the root/context table structure, the
|
||||
//! translation-enable and invalidation register sequences, and the fault drain. The
|
||||
//! core owns the domain table and the page-table walk; this file owns the registers.
|
||||
//!
|
||||
//! Register model (VT-d spec §10-11): offsets from the DRHD register base. The unit is
|
||||
//! programmed once at enable (root table + Translation Enable), then touched only for
|
||||
//! per-device context changes, per-domain invalidations, and fault draining. Interrupt
|
||||
//! remapping is deliberately left OFF (GCMD.IRE stays 0): with it off, upstream writes
|
||||
//! to 0xFEE0_0000-0xFEEF_FFFF are treated as interrupt requests and bypass second-level
|
||||
//! translation, so the existing MSI contract survives unchanged.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const architecture = @import("architecture");
|
||||
const log = @import("log.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// Register offsets from the unit's base.
|
||||
const reg_cap = 0x08; // Capability (64)
|
||||
const reg_ecap = 0x10; // Extended Capability (64)
|
||||
const reg_gcmd = 0x18; // Global Command (32, write-only)
|
||||
const reg_gsts = 0x1C; // Global Status (32, read-only)
|
||||
const reg_rtaddr = 0x20; // Root Table Address (64)
|
||||
const reg_ccmd = 0x28; // Context Command (64)
|
||||
const reg_fsts = 0x34; // Fault Status (32)
|
||||
|
||||
const gcmd_te: u32 = 1 << 31; // Translation Enable
|
||||
const gcmd_srtp: u32 = 1 << 30; // Set Root Table Pointer
|
||||
const gsts_tes: u32 = 1 << 31; // Translation Enable Status
|
||||
const gsts_rtps: u32 = 1 << 30; // Root Table Pointer Status
|
||||
|
||||
const cap_cm: u64 = 1 << 7; // Caching Mode
|
||||
const cap_sagaw_shift = 8; // Supported Adjusted Guest Address Widths, bits 12:8
|
||||
const cap_sagaw_39bit: u64 = 1 << 9; // 3-level
|
||||
const cap_sagaw_48bit: u64 = 1 << 10; // 4-level
|
||||
const cap_fro_shift = 24; // Fault-Recording Register Offset, bits 33:24 (×16)
|
||||
const cap_nfr_shift = 40; // Number of Fault Recording regs, bits 47:40 (+1)
|
||||
const ecap_coherent: u64 = 1 << 0; // hardware snoops CPU caches reading its structures
|
||||
const ecap_iro_shift = 8; // IOTLB Register Offset, bits 17:8 (×16)
|
||||
|
||||
const ccmd_icc: u64 = 1 << 63; // Invalidate Context-Cache
|
||||
const ccmd_cirg_global: u64 = @as(u64, 1) << 61; // global granularity
|
||||
const ccmd_cirg_device: u64 = @as(u64, 3) << 61; // device-selective
|
||||
|
||||
const iotlb_ivt: u64 = 1 << 63; // Invalidate IOTLB
|
||||
const iotlb_iirg_global: u64 = @as(u64, 1) << 60;
|
||||
const iotlb_iirg_domain: u64 = @as(u64, 2) << 60;
|
||||
const iotlb_dr: u64 = 1 << 49; // drain reads
|
||||
const iotlb_dw: u64 = 1 << 48; // drain writes
|
||||
|
||||
const fsts_ppf: u32 = 1 << 1; // Primary Pending Fault
|
||||
|
||||
// Second-level PTE bits.
|
||||
const slpte_read: u64 = 1 << 0;
|
||||
const slpte_write: u64 = 1 << 1;
|
||||
const slpte_page_size: u64 = 1 << 7; // a 2 MiB leaf (== iommu.huge_leaf_bit)
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
var register_base: usize = 0;
|
||||
var capabilities: u64 = 0;
|
||||
var extended_capabilities: u64 = 0;
|
||||
var coherent: bool = true; // ECAP.C — whether clflush is unnecessary
|
||||
var levels: u8 = 4;
|
||||
var context_aw: u64 = 2; // context-entry AW field (001=3-level, 010=4-level)
|
||||
var gcmd_shadow: u32 = 0; // sticky GCMD bits (TE etc.), for the write-only register
|
||||
|
||||
var root_table: u64 = 0; // physical base of the 256-entry root table
|
||||
var context_table: [256]u64 = .{0} ** 256; // per-bus context table physical, 0 = none
|
||||
|
||||
var fault_log_budget: u32 = 32; // rate-limit: log this many faults, then just count
|
||||
var faults_suppressed: u64 = 0;
|
||||
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*const volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write32(offset: usize, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, read caps, pick the address width. Returns the vtable, or
|
||||
/// null when the unit advertises no address width danos can drive.
|
||||
pub fn detect(info: platform.PlatformInformation) ?iommu.Backend {
|
||||
// Remap 16 KiB: FRCD and IOTLB registers can sit past the first page (CAP.FRO /
|
||||
// ECAP.IRO are 16-byte-unit offsets). Idempotent with the detection-time mapping.
|
||||
register_base = architecture.mapMmio(info.iommu_base, 16 * 1024, true);
|
||||
capabilities = read64(reg_cap);
|
||||
extended_capabilities = read64(reg_ecap);
|
||||
coherent = (extended_capabilities & ecap_coherent) != 0;
|
||||
|
||||
const sagaw = capabilities >> cap_sagaw_shift;
|
||||
if (sagaw & cap_sagaw_48bit != 0) {
|
||||
levels = 4;
|
||||
context_aw = 2; // 010b
|
||||
} else if (sagaw & cap_sagaw_39bit != 0) {
|
||||
levels = 3;
|
||||
context_aw = 1; // 001b
|
||||
} else {
|
||||
return null; // no width we build tables for
|
||||
}
|
||||
|
||||
root_table = allocZeroed() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = true,
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the root table and turn Translation Enable on. The core has already created
|
||||
/// and populated the RMRR domains (their context entries are live via `attach`), so at
|
||||
/// this instant every OTHER device's context entry is not-present and will fault — which
|
||||
/// for stale firmware bus-mastering is the desired evidence, not a bug.
|
||||
pub fn enable() void {
|
||||
write64(reg_rtaddr, root_table); // legacy mode (bits 11:10 = 00)
|
||||
setGlobalCommand(gcmd_srtp);
|
||||
spinStatus(gsts_rtps);
|
||||
globalInvalidate();
|
||||
setGlobalCommand(gcmd_te);
|
||||
spinStatus(gsts_tes);
|
||||
gcmd_shadow |= gcmd_te;
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
return (physical & address_mask) | slpte_read | slpte_write | (if (huge) slpte_page_size else 0);
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
_ = level;
|
||||
return (table_physical & address_mask) | slpte_read | slpte_write;
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & (slpte_read | slpte_write)) != 0;
|
||||
}
|
||||
|
||||
fn flushStructure(address: usize) void {
|
||||
if (coherent) return; // the unit snoops CPU caches; no flush needed (QEMU)
|
||||
asm volatile ("clflush (%[p])"
|
||||
:
|
||||
: [p] "r" (address),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
|
||||
// Lazily allocate this bus's context table and link it into the root table.
|
||||
if (context_table[bus] == 0) {
|
||||
const table = allocZeroed() orelse return;
|
||||
context_table[bus] = table;
|
||||
const root_entry = &tableAt(root_table)[@as(usize, bus) * 2]; // 16-byte entries
|
||||
root_entry.* = (table & address_mask) | 1; // present
|
||||
flushStructure(@intFromPtr(root_entry));
|
||||
}
|
||||
|
||||
const context = tableAt(context_table[bus]);
|
||||
const low = &context[@as(usize, devfn) * 2];
|
||||
const high = &context[@as(usize, devfn) * 2 + 1];
|
||||
high.* = (context_aw & 0x7) | (@as(u64, domain) << 8); // AW + DID
|
||||
low.* = (page_table_root & address_mask) | 1; // present, TT=00 (use second-level)
|
||||
flushStructure(@intFromPtr(high));
|
||||
flushStructure(@intFromPtr(low));
|
||||
|
||||
invalidateContextDevice(bdf, domain);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
if (context_table[bus] == 0) return;
|
||||
const context = tableAt(context_table[bus]);
|
||||
context[@as(usize, devfn) * 2] = 0; // not present
|
||||
context[@as(usize, devfn) * 2 + 1] = 0;
|
||||
flushStructure(@intFromPtr(&context[@as(usize, devfn) * 2]));
|
||||
invalidateContextDevice(bdf, 0);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_domain | iotlb_dr | iotlb_dw | (@as(u64, domain) << 32));
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const fsts = read32(reg_fsts);
|
||||
if (fsts & fsts_ppf == 0) return 0;
|
||||
|
||||
const fro = (capabilities >> cap_fro_shift) & 0x3FF;
|
||||
const nfr = ((capabilities >> cap_nfr_shift) & 0xFF) + 1;
|
||||
const frcd_base = @as(usize, @intCast(fro)) * 16;
|
||||
|
||||
var seen: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i < nfr) : (i += 1) {
|
||||
const off = frcd_base + i * 16;
|
||||
const high = read64(off + 8);
|
||||
if (high & (@as(u64, 1) << 63) == 0) continue; // F: no fault recorded here
|
||||
const low = read64(off);
|
||||
const address = low & ~@as(u64, 0xFFF);
|
||||
const source: u16 = @intCast(high & 0xFFFF);
|
||||
const reason: u8 = @intCast((high >> 32) & 0xFF);
|
||||
const is_read = (high >> 62) & 1; // T: 1 = read request
|
||||
logFault(source, address, reason, is_read == 1);
|
||||
write64(off + 8, @as(u64, 1) << 63); // RW1C: clear F
|
||||
seen += 1;
|
||||
}
|
||||
write32(reg_fsts, fsts); // clear PPF/PFO (write-1-to-clear)
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64, reason: u8, is_read: bool) void {
|
||||
if (fault_log_budget > 0) {
|
||||
fault_log_budget -= 1;
|
||||
log.print("DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=0x{x} write={d}\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
reason,
|
||||
@intFromBool(!is_read),
|
||||
});
|
||||
if (fault_log_budget == 0)
|
||||
log.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
} else {
|
||||
faults_suppressed += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- register helpers ----------------------------------------------------------------
|
||||
|
||||
fn setGlobalCommand(one_shot: u32) void {
|
||||
// GCMD is write-only: every write must carry the full sticky state plus the one-shot
|
||||
// bit being requested, or a set sticky bit (TE) would be cleared as a side effect.
|
||||
write32(reg_gcmd, gcmd_shadow | one_shot);
|
||||
}
|
||||
|
||||
fn spinStatus(bit: u32) void {
|
||||
var spins: u64 = 0;
|
||||
while (read32(reg_gsts) & bit == 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) {
|
||||
log.write("/system/kernel: WARNING VT-d status bit never set — translation may be incomplete\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn spin64(offset: usize, bit: u64) void {
|
||||
var spins: u64 = 0;
|
||||
while (read64(offset) & bit != 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) return;
|
||||
}
|
||||
}
|
||||
|
||||
fn globalInvalidate() void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_global);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn globalIotlb() void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_global | iotlb_dr | iotlb_dw);
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn invalidateContextDevice(bdf: u16, domain: u16) void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_device | (@as(u64, bdf) << 16) | domain);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
}
|
||||
|
||||
fn iotlbOffset() usize {
|
||||
const iro = (extended_capabilities >> ecap_iro_shift) & 0x3FF;
|
||||
return @as(usize, @intCast(iro)) * 16 + 8; // IOTLB register sits at IRO*16 + 8
|
||||
}
|
||||
|
||||
fn allocZeroed() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
@@ -0,0 +1,448 @@
|
||||
//! system/kernel/iommu.zig — vendor-neutral IOMMU core: per-device DMA translation
|
||||
//! domains over an Intel VT-d or AMD-Vi backend.
|
||||
//!
|
||||
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
||||
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
||||
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
||||
//! PCI function its own translation domain; a device reaches only the physical ranges
|
||||
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
||||
//! visible to it.
|
||||
//!
|
||||
//! Design:
|
||||
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
||||
//! physical address they program into hardware; a domain simply makes that same
|
||||
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
||||
//! driver's register-programming code is untouched.
|
||||
//! - **Vendor-neutral**: this file owns the domain table and a shared 512-entry
|
||||
//! page-table walker; a `Backend` vtable supplies the hardware specifics (VT-d in
|
||||
//! iommu-intel.zig, AMD-Vi in iommu-amd.zig) — the entry-bit encodings, the
|
||||
//! enable/invalidate register dances, and the fault drain.
|
||||
//! - **Fail-open**: when no IOMMU is found, `kind == .none` and every entry point is a
|
||||
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
||||
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
||||
//!
|
||||
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
||||
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const log = @import("log.zig");
|
||||
const intel = @import("iommu-intel.zig");
|
||||
const amd = @import("iommu-amd.zig");
|
||||
|
||||
const page_size: u64 = abi.page_size;
|
||||
const page_mask: u64 = page_size - 1;
|
||||
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||
|
||||
pub const Kind = enum { none, intel_vtd, amd_vi };
|
||||
|
||||
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
||||
pub const maximum_domains = 64;
|
||||
pub const invalid_domain: u16 = 0xFFFF;
|
||||
|
||||
/// The bit encodings and hardware operations a backend supplies to the shared core.
|
||||
/// Entry helpers build the raw page-table entries for the backend's format; the core
|
||||
/// walks the tree with them. The hardware ops act on a whole domain (identified by its
|
||||
/// hardware domain id = core index + 1) or device (by requester id / bdf).
|
||||
pub const Backend = struct {
|
||||
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
||||
levels: u8,
|
||||
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
||||
supports_huge_pages: bool,
|
||||
|
||||
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
||||
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
||||
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
||||
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
||||
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
||||
isPresent: *const fn (entry: u64) bool,
|
||||
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
||||
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
||||
flushStructure: *const fn (address: usize) void,
|
||||
|
||||
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
||||
/// context/device caches so the change takes effect.
|
||||
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
||||
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
||||
/// faults afterward.
|
||||
detach: *const fn (bdf: u16) void,
|
||||
/// Invalidate cached translations for `domain` (after a map or unmap).
|
||||
invalidateDomain: *const fn (domain: u16) void,
|
||||
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
||||
/// count seen this call.
|
||||
faultDrain: *const fn () usize,
|
||||
};
|
||||
|
||||
const Domain = struct {
|
||||
in_use: bool = false,
|
||||
owner: u32 = 0, // task that owns the attached device
|
||||
bdf: u16 = 0, // requester id of the attached device
|
||||
page_table_root: u64 = 0, // physical address of the top-level table
|
||||
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
||||
};
|
||||
|
||||
var kind: Kind = .none;
|
||||
var backend: Backend = undefined;
|
||||
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
||||
|
||||
pub fn kindOf() Kind {
|
||||
return kind;
|
||||
}
|
||||
pub fn enabled() bool {
|
||||
return kind != .none;
|
||||
}
|
||||
|
||||
/// Detect the IOMMU, pick a backend, pre-map firmware reserved regions, and enable
|
||||
/// translation. Fail-open (kind stays .none) when no unit exists — the caller logs the
|
||||
/// posture. Must run after platform discovery and before any user process starts.
|
||||
pub fn init() void {
|
||||
const info = platform.platformInformation();
|
||||
if (!info.iommu_present) {
|
||||
kind = .none;
|
||||
return;
|
||||
}
|
||||
// Pick the backend by vendor. A present-but-unusable unit stays fail-open with a
|
||||
// logged reason rather than half-enabling.
|
||||
if (info.iommu_is_amd) {
|
||||
if (amd.detect(info)) |be| {
|
||||
backend = be;
|
||||
kind = .amd_vi;
|
||||
} else {
|
||||
kind = .none;
|
||||
log.write("/system/kernel: WARNING AMD-Vi present but unusable — staying fail-open\n");
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
if (intel.detect(info)) |be| {
|
||||
backend = be;
|
||||
kind = .intel_vtd;
|
||||
} else {
|
||||
kind = .none;
|
||||
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// The translation structures start empty: every device is denied until its driver
|
||||
// claims it (confineDevice gives it a private domain). PCI functions are enumerated
|
||||
// post-boot by the ring-3 pci-bus driver, so there is nothing to attach at init.
|
||||
if (kind == .amd_vi) amd.enable() else intel.enable();
|
||||
logEnabled(info);
|
||||
}
|
||||
|
||||
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||
/// exactly the domains it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
||||
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
||||
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
||||
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
||||
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
||||
/// IOMMU exists (fail-open).
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (kind == .none) return true;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
const domain = domainCreate(owner, bdf) orelse return false;
|
||||
|
||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||
const info = platform.platformInformation();
|
||||
var i: usize = 0;
|
||||
while (i < info.rmrr_count) : (i += 1) {
|
||||
if (info.rmrr[i].bdf == bdf)
|
||||
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
||||
}
|
||||
|
||||
attachDevice(domain, bdf);
|
||||
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The confined record for `device_id`, or null if the device is not confined.
|
||||
fn confinedOf(device_id: u64) ?*Confined {
|
||||
if (device_id >= confined.len) return null;
|
||||
const c = &confined[@intCast(device_id)];
|
||||
return if (c.active) c else null;
|
||||
}
|
||||
|
||||
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
|
||||
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
|
||||
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
|
||||
if (kind == .none) return true;
|
||||
const c = confinedOf(device_id) orelse return false;
|
||||
return map(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
|
||||
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
const c = confinedOf(device_id) orelse return;
|
||||
unmap(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
|
||||
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
|
||||
/// before the frames return to pmm: a device translating to a reallocated frame is the
|
||||
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
|
||||
/// domain other than its owner's.
|
||||
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
if (kind == .none) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) {
|
||||
detachDevice(c.bdf);
|
||||
domainDestroy(c.domain);
|
||||
c.* = .{};
|
||||
}
|
||||
}
|
||||
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
||||
}
|
||||
|
||||
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
||||
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
||||
if (kind == .none) return 0; // fail-open: a dummy id the no-op ops ignore
|
||||
for (&domains, 0..) |*d, index| {
|
||||
if (d.in_use) continue;
|
||||
const root = allocTable() orelse return null;
|
||||
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
||||
return @intCast(index);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
||||
/// (detach first).
|
||||
pub fn domainDestroy(domain: u16) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
freeTables(d.page_table_root, backend.levels);
|
||||
d.* = .{};
|
||||
}
|
||||
|
||||
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
||||
pub fn attachDevice(domain: u16, bdf: u16) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
d.bdf = bdf;
|
||||
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
||||
}
|
||||
|
||||
/// Return `bdf`'s device to not-present + invalidate.
|
||||
pub fn detachDevice(bdf: u16) void {
|
||||
if (kind == .none) return;
|
||||
backend.detach(bdf);
|
||||
}
|
||||
|
||||
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
||||
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
||||
/// caching-mode and free otherwise.
|
||||
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
||||
if (kind == .none) return true;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return false;
|
||||
if (!mapRange(d.page_table_root, physical, len)) return false;
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
||||
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
||||
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
||||
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
unmapRange(d.page_table_root, physical, len);
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
}
|
||||
|
||||
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
||||
/// IOMMU test case and opportunistically after a device detaches.
|
||||
pub fn faultDrain() usize {
|
||||
if (kind == .none) return 0;
|
||||
return backend.faultDrain();
|
||||
}
|
||||
|
||||
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
||||
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
||||
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
||||
if (kind == .none) return virtual;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return null;
|
||||
var table = d.page_table_root;
|
||||
var level = backend.levels;
|
||||
while (level > 1) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(virtual, level)];
|
||||
if (!backend.isPresent(entry)) return null;
|
||||
if (level == 2 and isHugeLeaf(entry))
|
||||
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
||||
if (!backend.isPresent(leaf)) return null;
|
||||
return (leaf & address_mask) | (virtual & page_mask);
|
||||
}
|
||||
|
||||
// --- the shared page-table walker -----------------------------------------------------
|
||||
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
||||
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
fn indexAt(virtual: u64, level: u8) usize {
|
||||
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
||||
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
||||
return @intCast((virtual >> shift) & 0x1FF);
|
||||
}
|
||||
|
||||
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
||||
/// physical base. null on out-of-memory.
|
||||
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
||||
const entry = entry_ptr.*;
|
||||
if (backend.isPresent(entry)) return entry & address_mask;
|
||||
const table = allocTable() orelse return null;
|
||||
backend.flushStructure(@intFromPtr(tableAt(table)));
|
||||
entry_ptr.* = backend.makeTable(table, level);
|
||||
backend.flushStructure(@intFromPtr(entry_ptr));
|
||||
return table;
|
||||
}
|
||||
|
||||
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
||||
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
||||
// cases without a separate superpage path per backend.
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
if (!mapOne(root, addr, huge)) return false;
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
||||
table = descend(entry_ptr, level) orelse return false;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
return true;
|
||||
}
|
||||
|
||||
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
unmapOne(root, addr, huge);
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
}
|
||||
|
||||
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(addr, level)];
|
||||
if (!backend.isPresent(entry)) return; // nothing mapped here
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = 0;
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
}
|
||||
|
||||
/// Post-order free of a domain's whole table tree.
|
||||
fn freeTables(root: u64, level: u8) void {
|
||||
if (level > 1) {
|
||||
const table = tableAt(root);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) {
|
||||
const entry = table[i];
|
||||
if (!backend.isPresent(entry)) continue;
|
||||
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
||||
if (level == 2 and isHugeLeaf(entry)) continue;
|
||||
freeTables(entry & address_mask, level - 1);
|
||||
}
|
||||
}
|
||||
pmm.free(root);
|
||||
}
|
||||
|
||||
fn isHugeLeaf(entry: u64) bool {
|
||||
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
||||
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
||||
// pointer" at level 2, which huge leaves are by construction.
|
||||
return entry & huge_leaf_bit != 0;
|
||||
}
|
||||
|
||||
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
||||
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
||||
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
||||
pub const huge_leaf_bit: u64 = 1 << 7;
|
||||
|
||||
fn hardwareId(domain: u16) u16 {
|
||||
return domain + 1; // id 0 is reserved by both architectures
|
||||
}
|
||||
|
||||
fn logEnabled(info: platform.PlatformInformation) void {
|
||||
if (kind == .amd_vi) {
|
||||
log.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
||||
log.print(" levels : {d} (48-bit)\n", .{backend.levels});
|
||||
return;
|
||||
}
|
||||
log.write("/system/kernel: iommu online (Intel VT-d)\n");
|
||||
log.print(" version : 0x{x}\n", .{info.iommu_version});
|
||||
log.print(" agaw : {d} levels\n", .{backend.levels});
|
||||
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
||||
if (info.rmrr_skipped > 0)
|
||||
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
||||
if (info.iommu_extra_units > 0)
|
||||
log.print(" units : WARNING {d} other DRHD(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
||||
}
|
||||
@@ -154,6 +154,66 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||
pub const handle_kind_endpoint: u8 = 0;
|
||||
pub const handle_kind_shared_memory: u8 = 1;
|
||||
pub const handle_kind_dma_region: u8 = 2;
|
||||
|
||||
/// A DMA buffer's delegation token: names the `pages` contiguous frames at `phys` that
|
||||
/// `dma_alloc` handed its `owner`, and can be passed across processes as a capability so
|
||||
/// the driver that owns a device can bind it into that device's IOMMU domain
|
||||
/// (`dma_bind`). Unlike `SharedMemoryObject` this does NOT own the frames — the
|
||||
/// allocating address space still does, and frees them on `dma_free` or teardown — so
|
||||
/// this is a pure token: `dead` is set when the allocator frees the region, after which
|
||||
/// a stale handle can no longer bind it. `refcount` counts the allocator's registry
|
||||
/// entry plus every outstanding handle; the object is freed when the last drops.
|
||||
pub const DmaRegionObject = struct {
|
||||
refcount: u32 = 1,
|
||||
phys: u64,
|
||||
pages: usize,
|
||||
owner: u32,
|
||||
dead: bool = false,
|
||||
};
|
||||
|
||||
/// Create a DMA-region token for `pages` frames at `phys` owned by task `owner`. The
|
||||
/// frames are already allocated and mapped by the caller; this only wraps them for
|
||||
/// delegation. null if the heap is out of room.
|
||||
pub fn createDmaRegion(phys: u64, pages: usize, owner: u32) ?*DmaRegionObject {
|
||||
const region = heap.allocator().create(DmaRegionObject) catch return null;
|
||||
region.* = .{ .phys = phys, .pages = pages, .owner = owner };
|
||||
return region;
|
||||
}
|
||||
|
||||
/// Drop a DMA-region reference; free the token when the last (registry + handles) goes.
|
||||
/// Never frees frames — the allocator owns those.
|
||||
pub fn dropDmaRegionReference(region: *DmaRegionObject) void {
|
||||
if (region.refcount > 1) {
|
||||
region.refcount -= 1;
|
||||
} else {
|
||||
heap.allocator().destroy(region);
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolve a handle to its DMA-region token, or null if out of range, unused, or a
|
||||
/// different kind.
|
||||
pub fn resolveDmaRegion(t: *Task, h: u64) ?*DmaRegionObject {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_dma_region) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Install a DMA-region handle in task `t`'s table (the slot owns a reference).
|
||||
pub fn installDmaRegionHandle(t: *Task, region: *DmaRegionObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_dma_region, .ptr = @ptrCast(region) });
|
||||
}
|
||||
|
||||
/// Drop the handle at slot `h` of task `t` (handle_close): release its reference and
|
||||
/// free the slot. Returns 0 or -EBADF.
|
||||
pub fn closeHandle(t: *Task, h: u64) i64 {
|
||||
if (h >= t.handles.len) return -EBADF;
|
||||
const entry = t.handles[@intCast(h)] orelse return -EBADF;
|
||||
dropEntry(entry);
|
||||
t.handles[@intCast(h)] = null;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||
@@ -307,6 +367,10 @@ fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||
const s: *SharedMemoryObject = @ptrCast(@alignCast(entry.ptr));
|
||||
s.refcount += 1;
|
||||
},
|
||||
handle_kind_dma_region => {
|
||||
const r: *DmaRegionObject = @ptrCast(@alignCast(entry.ptr));
|
||||
r.refcount += 1;
|
||||
},
|
||||
else => return -EBADF,
|
||||
}
|
||||
const handle = installEntry(to, entry);
|
||||
@@ -574,6 +638,7 @@ fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shared_memory => dropSharedMemoryReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_dma_region => dropDmaRegionReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
@@ -215,6 +216,12 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Bring up DMA translation: build the IOMMU domains and enable it (or record
|
||||
// fail-open when no unit exists). Must run before any driver claims a device —
|
||||
// an unclaimed device's DMA is blocked once translation is on. Logs its own
|
||||
// enable block; the fail-open posture is stated in the platform block below.
|
||||
iommu.init();
|
||||
|
||||
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||
const pw = platform.powerInformation();
|
||||
@@ -272,6 +279,11 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
// The DMA-isolation posture, stated plainly at every boot. When a unit exists,
|
||||
// iommu.init already logged its enable block above; here we only state the
|
||||
// fail-open case, so a boot without the line is a boot with translation on.
|
||||
if (!iommu.enabled())
|
||||
log.write(" iommu : none present - DMA fail-open (unisolated)\n");
|
||||
} else |err| {
|
||||
log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
|
||||
@@ -33,6 +33,7 @@ const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const vfs = @import("vfs.zig");
|
||||
const log = @import("log.zig");
|
||||
@@ -250,6 +251,10 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.fs_node => systemFsNode(state),
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.iommu_fault_drain => systemIommuFaultDrain(state),
|
||||
.dma_bind => systemDmaBind(state),
|
||||
.dma_unbind => systemDmaUnbind(state),
|
||||
.handle_close => systemHandleClose(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shared_memory_create => systemSharedMemoryCreate(state),
|
||||
.shared_memory_map => systemSharedMemoryMap(state),
|
||||
@@ -390,6 +395,21 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
const claim_flags = sync.enter();
|
||||
defer sync.leave(claim_flags);
|
||||
if (devices_broker.claim(device_id, scheduler.current().id)) {
|
||||
// Confine the device's DMA before the driver can program it: a PCI function
|
||||
// becomes reachable to the IOMMU only once claimed (until now its DMA is
|
||||
// blocked). A claim that cannot be confined must not stand — roll it back —
|
||||
// since the whole point is that claiming a DMA device is no longer equivalent
|
||||
// to ring 0. No-op when no IOMMU exists (fail-open).
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
const owner = scheduler.current().id;
|
||||
if (!iommu.confineDevice(device_id, bdf, owner)) {
|
||||
_ = devices_broker.unclaim(device_id, owner);
|
||||
return fail(state);
|
||||
}
|
||||
// Bind the buffers this task allocated before claiming the device (a driver
|
||||
// that dma_alloc'd its rings, then claimed the controller).
|
||||
dmaBindOwnerRegionsInto(owner, device_id);
|
||||
}
|
||||
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||
// so the kernel and the service don't scribble over each other's pixels. The
|
||||
// claim releases (and the console resumes) automatically if the service dies;
|
||||
@@ -497,6 +517,137 @@ fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
|
||||
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
|
||||
/// until PAT is programmed. See docs/driver-model.md (M14).
|
||||
// --- DMA-region registry ---------------------------------------------------------------
|
||||
// Every dma_alloc'd region is tracked here so it can be (a) auto-bound into the devices
|
||||
// its owner claims, (b) delegated across processes as a capability and bound into a
|
||||
// device's IOMMU domain by dma_bind, and (c) unmapped from every domain before its frames
|
||||
// return to the allocator. Only *shareable* regions carry a heap `object` (the delegation
|
||||
// token); a driver's private rings are tracked without one. All access under the big lock.
|
||||
const DmaRegistryEntry = struct {
|
||||
active: bool = false,
|
||||
object: ?*ipc.DmaRegionObject = null,
|
||||
physical: u64 = 0,
|
||||
len: u64 = 0,
|
||||
owner: u32 = 0,
|
||||
};
|
||||
const maximum_dma_regions = 256;
|
||||
var dma_registry: [maximum_dma_regions]DmaRegistryEntry = .{DmaRegistryEntry{}} ** maximum_dma_regions;
|
||||
|
||||
fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (!e.active) {
|
||||
e.* = .{ .active = true, .object = object, .physical = physical, .len = len, .owner = owner };
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire the region based at `physical` owned by `owner`: unmap it from every device
|
||||
/// domain (before the frames are freed), mark its token dead so a stale downstream handle
|
||||
/// can no longer bind it, drop the registry's reference, and clear the slot.
|
||||
fn dmaRegistryRemove(owner: u32, physical: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner and e.physical == physical) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire every region owned by `owner` (task death) — same discipline as a per-region
|
||||
/// free, run before the address space is torn down and its DMA frames reclaimed.
|
||||
fn dmaRegistryReleaseOwner(owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Bind every region `owner` allocated into the domain of the device it just claimed
|
||||
/// (its rings allocated before the claim). Regions allocated *after* the claim are bound
|
||||
/// by dma_alloc's own auto-bind.
|
||||
fn dmaBindOwnerRegionsInto(owner: u32, device_id: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) _ = iommu.mapForDevice(device_id, e.physical, e.len);
|
||||
}
|
||||
}
|
||||
|
||||
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
|
||||
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
|
||||
/// its device faulted can force the fault records to be logged now rather than waiting
|
||||
/// for the next device-release drain. Harmless without an IOMMU (returns 0).
|
||||
fn systemIommuFaultDrain(state: *architecture.CpuState) void {
|
||||
const flags = sync.enter();
|
||||
const count = iommu.faultDrain();
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, count);
|
||||
}
|
||||
|
||||
/// Resolve a capability handle the caller holds to the physical range it names — either
|
||||
/// a DMA-region token or a shared-memory object (both are bindable buffers). null if the
|
||||
/// handle is neither, or names a region already freed by its allocator.
|
||||
fn bindableRange(t: *scheduler.Task, handle: u64) ?struct { physical: u64, len: u64 } {
|
||||
if (ipc.resolveDmaRegion(t, handle)) |region| {
|
||||
if (region.dead) return null;
|
||||
return .{ .physical = region.phys, .len = region.pages * page_size };
|
||||
}
|
||||
if (ipc.resolveSharedMemory(t, handle)) |shared| {
|
||||
return .{ .physical = shared.phys, .len = shared.pages * page_size };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// dma_bind(device_id, handle) -> 0/-errno: map the buffer named by `handle` into the
|
||||
/// claimed device's IOMMU domain. The caller must own the device (the mmio_map gate) and
|
||||
/// hold the handle. Idempotent: re-binding is a harmless success, so a driver may re-bind
|
||||
/// after a restart without tracking what it already bound. Success (no-op) without an IOMMU.
|
||||
fn systemDmaBind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
if (!iommu.mapForDevice(device_id, range.physical, range.len)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// dma_unbind(device_id, handle) -> 0/-errno: unmap a previously bound buffer from the
|
||||
/// device's domain and invalidate.
|
||||
fn systemDmaUnbind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
iommu.unmapForDevice(device_id, range.physical, range.len);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// handle_close(handle) -> 0/-errno: drop one capability handle and free its slot.
|
||||
fn systemHandleClose(state: *architecture.CpuState) void {
|
||||
const handle = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (ipc.closeHandle(t, handle) < 0) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
@@ -543,8 +694,33 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
// Register the region and bind it into every device this task already drives (its
|
||||
// own buffers reach its own devices). When `dma_shareable` is set, also wrap it in a
|
||||
// capability token and return a handle so it can be delegated to another driver and
|
||||
// dma_bound there. All under the lock (the registry + IOMMU tables are shared).
|
||||
var handle: u64 = abi.no_cap;
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
defer sync.leave(lock_flags);
|
||||
var object: ?*ipc.DmaRegionObject = null;
|
||||
if (flags & abi.dma_shareable != 0) {
|
||||
if (ipc.createDmaRegion(phys, pages, t.id)) |region| {
|
||||
const h = ipc.installDmaRegionHandle(t, region);
|
||||
if (h >= 0) {
|
||||
region.refcount += 1; // the handle's reference (registry holds the first)
|
||||
handle = @intCast(h);
|
||||
object = region;
|
||||
} else {
|
||||
ipc.dropDmaRegionReference(region); // no table slot; drop it
|
||||
}
|
||||
}
|
||||
}
|
||||
dmaRegistryAdd(object, phys, pages * page_size, t.id);
|
||||
iommu.mapRegionForOwner(t.id, phys, pages * page_size);
|
||||
}
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
architecture.setSystemCallResult3(state, handle); // capability handle (no_cap unless shareable)
|
||||
}
|
||||
|
||||
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
@@ -560,6 +736,16 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
// Retire the region — unmap it from every device domain and mark its token dead —
|
||||
// BEFORE any frame returns to the allocator, so no device can still translate to a
|
||||
// reallocated frame. dma_alloc's frames are contiguous, so the base names the region.
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
if (architecture.translate(t.address_space, base_v)) |base_phys|
|
||||
dmaRegistryRemove(t.id, base_phys);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base_v + i * page_size;
|
||||
// Per-page lock hold: the translate/unmap walks the shared page tables
|
||||
@@ -991,6 +1177,14 @@ pub var fault_kill_count: u64 = 0;
|
||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
recordExitLocked(t);
|
||||
irq.releaseOwner(t.id);
|
||||
// Detach the task's devices from their IOMMU domains BEFORE the broker clears the
|
||||
// claims (the detach reads ownership) and before the address space is torn down and
|
||||
// its DMA frames return to the allocator — a device must stop translating to a frame
|
||||
// before that frame can be handed to someone else. Then retire the buffers this task
|
||||
// allocated, unmapping them from any *other* driver's domain they were granted into,
|
||||
// before those frames are freed too.
|
||||
iommu.releaseAllOwnedBy(t.id);
|
||||
dmaRegistryReleaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
// If that dropped the framebuffer claim (this task was the display service), let the
|
||||
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
|
||||
|
||||
@@ -149,7 +149,10 @@ pub const maximum_task_name = abi.maximum_process_name;
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
// Raised from 16 with DMA-region capabilities: a driver now holds its per-device
|
||||
// channel endpoints plus received DMA-region handles (a storage driver forwards several
|
||||
// buffer caps), and repeated cap-passing consumes slots until handle_close.
|
||||
pub const ipc_maximum_handles = 32;
|
||||
|
||||
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||
|
||||
+109
-6
@@ -18,6 +18,7 @@ const wall_clock = @import("wall-clock.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const platform = @import("platform");
|
||||
const pmm = @import("pmm.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
@@ -212,6 +213,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
pciScanTest(boot_information);
|
||||
} else if (eql(case, "pci-caps")) {
|
||||
pciCapsTest(boot_information);
|
||||
} else if (eql(case, "iommu-fault")) {
|
||||
iommuFaultTest(boot_information);
|
||||
} else if (eql(case, "acpi-parse")) {
|
||||
acpiParseTest(boot_information);
|
||||
} else if (eql(case, "acpi-report")) {
|
||||
@@ -1164,6 +1169,11 @@ fn dmaTest() void {
|
||||
log("DANOS-TEST-BEGIN: dma\n", .{});
|
||||
const base_free = pmm.stats().free_frames;
|
||||
|
||||
// This case boots WITHOUT an IOMMU device, so it is the explicit witness of the
|
||||
// fail-open posture: no unit found, and the kernel said so at boot (the harness
|
||||
// asserts the boot line; this check pins the recorded state to it).
|
||||
check("no IOMMU present: DMA runs fail-open", !platform.platformInformation().iommu_present);
|
||||
|
||||
// A contiguous run: aligned, and it consumed exactly that many frames.
|
||||
const frames = 4;
|
||||
const phys = pmm.allocContiguous(frames, ~@as(u64, 0)) orelse {
|
||||
@@ -1234,16 +1244,41 @@ fn msiTest() void {
|
||||
|
||||
/// IOMMU (M16): with an emulated VT-d unit present (the harness boots this case with
|
||||
/// `-device intel-iommu`), danos must find it in the ACPI DMAR table, map its register
|
||||
/// block, and read back a real version. This is *detection*, the honest first step —
|
||||
/// no translation domains are programmed yet, so DMA is still unprotected; enforcement
|
||||
/// lands with the first DMA driver (docs/driver-model.md M16).
|
||||
/// block, and read back a real version. Detection was M16's honest first step; the
|
||||
/// IOVA-enforcement track extends this case milestone by milestone (translation
|
||||
/// enabled, then per-device domains) — see the plan in docs and the fail-open witness
|
||||
/// in dmaTest.
|
||||
fn iommuTest() void {
|
||||
log("DANOS-TEST-BEGIN: iommu\n", .{});
|
||||
const pinfo = platform.platformInformation();
|
||||
check("IOMMU found in the DMAR table", pinfo.iommu_present);
|
||||
check("VT-d unit has a register base", pinfo.iommu_base != 0);
|
||||
check("VT-d version register reads back nonzero (real, mappable unit)", pinfo.iommu_version != 0);
|
||||
check("IOMMU found in the firmware tables", pinfo.iommu_present);
|
||||
check("IOMMU unit has a register base", pinfo.iommu_base != 0);
|
||||
// The VT-d version register is a live-unit sanity check; AMD-Vi (from IVRS) records
|
||||
// no version, so gate it on the vendor.
|
||||
if (!pinfo.iommu_is_amd)
|
||||
check("VT-d version register reads back nonzero (real, mappable unit)", pinfo.iommu_version != 0);
|
||||
log("DANOS-IOMMU: base=0x{x} version=0x{x} capabilities=0x{x}\n", .{ pinfo.iommu_base, pinfo.iommu_version, pinfo.iommu_capabilities });
|
||||
|
||||
// Translation was enabled at boot (kernel.zig: iommu.init before any driver claims
|
||||
// a device). The blanket domain keeps every device identity-mapped, so DMA still
|
||||
// works, but the unit is live — and with only the boot-time mappings present, no
|
||||
// device should have faulted yet.
|
||||
check("IOMMU enabled (translation on)", iommu.enabled());
|
||||
check("no spurious translation faults at idle", iommu.faultDrain() == 0);
|
||||
|
||||
// A scratch domain proves the walker + invalidation path end to end: create it,
|
||||
// identity-map a page, and confirm the mapping resolves; then unmap and destroy.
|
||||
if (iommu.domainCreate(0, 0)) |scratch| {
|
||||
const scratch_phys: u64 = 0x0010_0000; // 1 MiB, page-aligned
|
||||
check("map into a scratch domain succeeds", iommu.map(scratch, scratch_phys, abi.page_size));
|
||||
check("scratch domain resolves the mapping", iommu.translationOf(scratch, scratch_phys) == scratch_phys);
|
||||
iommu.unmap(scratch, scratch_phys, abi.page_size);
|
||||
check("scratch domain drops the mapping", iommu.translationOf(scratch, scratch_phys) == null);
|
||||
iommu.domainDestroy(scratch);
|
||||
} else {
|
||||
check("scratch domain allocated", false);
|
||||
}
|
||||
log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base});
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -2388,6 +2423,74 @@ fn deviceListTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The driver-side PCI library against a real function: the pci-caps QEMU case adds an
|
||||
/// e1000e NIC no danos driver claims; the pci-cap-test fixture claims it and exercises
|
||||
/// header accessors, command bits, the capability walks, MSI programming (the first
|
||||
/// driver-side `msi_bind` use), the MSI-X table, power state, and FLR. The kernel side
|
||||
/// only spawns the manager (which spawns pci-bus itself) and the fixture; the substance
|
||||
/// is asserted by the harness on the fixture's own serial lines.
|
||||
fn pciCapsTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: pci-caps\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
// Plain mode — no restart drill, whose kill would race the fixture's claim.
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("pci-cap-test spawned", spawnNamed(rd, "pci-cap-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// IOMMU enforcement, the negative proof: boot with VT-d on and an unclaimed e1000e.
|
||||
/// The manager spawns pci-bus, the fixture claims the NIC and fires a DMA at an
|
||||
/// unmapped page; the unit must fault it and the system survive. Substance is asserted
|
||||
/// by the harness on the kernel's DANOS-IOMMU-FAULT line and the fixture's markers.
|
||||
fn iommuFaultTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: iommu-fault\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
check("IOMMU enabled for the enforcement test", iommu.enabled());
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("iommu-fault-test spawned", spawnNamed(rd, "iommu-fault-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||
|
||||
@@ -214,13 +214,15 @@ fn onInit(endpoint: ipc.Handle) bool {
|
||||
while (i < registered_count) : (i += 1) {
|
||||
const entry = registered[i];
|
||||
const hid = entry.hid[0..entry.hid_len];
|
||||
// The devices.csv columns (bus=acpi, hid) then the human-readable name — a
|
||||
// would-be /etc/devices.csv row read straight off the boot log.
|
||||
const desc = acpi_ids.description(hid);
|
||||
if (desc.len != 0)
|
||||
std.log.info("reported {s} (device {d}, {d} resources) — {s}", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
std.log.info("device {d} bus=acpi hid={s} — {s} ({d} resources)", .{ entry.device_id, hid, desc, entry.resource_count })
|
||||
else
|
||||
std.log.info("reported {s} (device {d}, {d} resources)", .{ hid, entry.device_id, entry.resource_count });
|
||||
std.log.info("device {d} bus=acpi hid={s} ({d} resources)", .{ entry.device_id, hid, entry.resource_count });
|
||||
if (manager) |h| {
|
||||
var report = device_manager_protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
var report = device_manager_protocol.ChildAdded{ .bus = @intFromEnum(device_manager_protocol.BusKind.acpi), .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||
|
||||
@@ -23,86 +23,66 @@ const service = @import("service");
|
||||
const time = @import("time");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const pci_class = @import("pci-class");
|
||||
const usb_ids = @import("usb-ids");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const registry = @import("device-registry");
|
||||
const fs = @import("file-system");
|
||||
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||
const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.serial_bus),
|
||||
.subclass = @intFromEnum(pci_class.serial_bus.SubClass.usb),
|
||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||
});
|
||||
// --- the device registry ------------------------------------------------------
|
||||
// Driver matching is data-driven and authoritative: /etc/devices.csv (parsed by
|
||||
// the device-registry module) names, per bus, which driver binds a reported
|
||||
// device, the most-specific match winning. There is no compiled-in fallback — a
|
||||
// device no row matches goes unbound and is logged. This retired the hand-kept
|
||||
// pciDriverForIdentity / hidDriverFor / usbDriverForIdentity switch tables
|
||||
// (docs/device-manager.md: "matching stays code until the third bus").
|
||||
|
||||
/// The PCI class triple of a virtio-gpu — Display Controller / Other (0x80) / 0. The class
|
||||
/// alone cannot tell it from any other display/other function, so the driver re-confirms
|
||||
/// vendor 0x1AF4 / device 0x1050 from config space once spawned; this only gets it spawned.
|
||||
const virtio_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.display),
|
||||
.subclass = 0x80, // "Other" — no named SubClass member (PCI convention)
|
||||
.prog_if = 0,
|
||||
});
|
||||
/// The CSV bytes, held for the life of the process because the parsed rules'
|
||||
/// string fields (hid, driver) slice into this buffer.
|
||||
var registry_source: [8192]u8 = undefined;
|
||||
var registry_rules: [64]registry.Rule = undefined;
|
||||
var registry_count: usize = 0;
|
||||
|
||||
|
||||
const vga_compatible_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.display),
|
||||
.subclass = @intFromEnum(pci_class.display.SubClass.vga_compatible),
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||
/// several identical controllers — one driver instance per reported device,
|
||||
/// its registered id as argv[1].
|
||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
return switch (identity) {
|
||||
xhci_pci_class => "/system/drivers/usb-xhci-bus",
|
||||
vga_compatible_gpu_pci_class => "/system/drivers/display",
|
||||
virtio_gpu_pci_class => "/system/drivers/virtio-gpu",
|
||||
else => null,
|
||||
/// Read and parse /etc/devices.csv once at boot. The file lives in the initial
|
||||
/// ramdisk, which the kernel serves directly — no filesystem service need be up
|
||||
/// (fat is spawned after the manager), so this is a plain fs.open + read.
|
||||
fn loadRegistry() void {
|
||||
var file = fs.open("/etc/devices.csv", .{}) orelse {
|
||||
_ = logging.write("/system/services/device-manager: /etc/devices.csv missing — nothing will match\n");
|
||||
return;
|
||||
};
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < registry_source.len) {
|
||||
const n = file.read(registry_source[used..]) orelse break;
|
||||
if (n == 0) break;
|
||||
used += n;
|
||||
}
|
||||
const result = registry.parse(registry_source[0..used], ®istry_rules);
|
||||
registry_count = result.count;
|
||||
if (result.malformed != 0) std.log.info("/etc/devices.csv: {d} malformed line(s) skipped", .{result.malformed});
|
||||
if (result.truncated) _ = logging.write("/system/services/device-manager: /etc/devices.csv has more rules than the table holds\n");
|
||||
std.log.info("/etc/devices.csv: {d} rule(s) loaded", .{registry_count});
|
||||
}
|
||||
|
||||
/// The driver that serves a *reported* ACPI device by its `_HID` (M20.3:
|
||||
/// ps2-bus now binds the PS/2 nodes the acpi service reports, not boot-snapshot
|
||||
/// nodes the kernel used to build). ps2-bus is a singleton that finds both its
|
||||
/// devices by hid once spawned, so keyboard and mouse map to the same name.
|
||||
fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
||||
if (std.mem.eql(u8, hid, "PNP0303")) return "/system/drivers/ps2-bus"; // PS/2 keyboard
|
||||
if (std.mem.eql(u8, hid, "PNP0F13")) return "/system/drivers/ps2-bus"; // PS/2 mouse
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The driver that serves a *reported* USB interface by its (class, subclass,
|
||||
/// protocol) triple — the third bus after PCI and ACPI (docs/device-manager.md:
|
||||
/// matching stays code until the third bus). The xHCI bus driver reports each
|
||||
/// interface with this packed triple as its identity; the matched class driver is
|
||||
/// spawned with the interface's registered id as argv[1], which it presents to the
|
||||
/// bus driver to open the device.
|
||||
fn usbDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
const keyboard = comptime usb_ids.packTriple(
|
||||
@intFromEnum(usb_ids.Class.hid),
|
||||
@intFromEnum(usb_ids.hid.SubClass.boot),
|
||||
@intFromEnum(usb_ids.hid.Protocol.keyboard),
|
||||
);
|
||||
const mouse = comptime usb_ids.packTriple(
|
||||
@intFromEnum(usb_ids.Class.hid),
|
||||
@intFromEnum(usb_ids.hid.SubClass.boot),
|
||||
@intFromEnum(usb_ids.hid.Protocol.mouse),
|
||||
);
|
||||
const storage = comptime usb_ids.packTriple(
|
||||
@intFromEnum(usb_ids.Class.mass_storage),
|
||||
@intFromEnum(usb_ids.mass_storage.SubClass.scsi),
|
||||
@intFromEnum(usb_ids.mass_storage.Protocol.bulk_only),
|
||||
);
|
||||
return switch (identity) {
|
||||
keyboard => "/system/drivers/usb-hid-keyboard",
|
||||
mouse => "/system/drivers/usb-hid-mouse",
|
||||
storage => "/system/drivers/usb-storage",
|
||||
else => null,
|
||||
/// Build a registry Identity from a bus driver's report: the bus it named, the
|
||||
/// class triple unpacked from `identity` (0xCCSSPP — the same packing for a PCI
|
||||
/// class code and a USB class triple), the widened numeric ids, and the ACPI hid.
|
||||
fn identityFromReport(report: device_manager_protocol.ChildAdded) registry.Identity {
|
||||
const bus: registry.Bus = switch (report.bus) {
|
||||
@intFromEnum(device_manager_protocol.BusKind.pci) => .pci,
|
||||
@intFromEnum(device_manager_protocol.BusKind.usb) => .usb,
|
||||
@intFromEnum(device_manager_protocol.BusKind.acpi) => .acpi,
|
||||
else => .unknown,
|
||||
};
|
||||
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
||||
return .{
|
||||
.bus = bus,
|
||||
.base = @truncate(report.identity >> 16),
|
||||
.subclass = @truncate(report.identity >> 8),
|
||||
.prog_if = @truncate(report.identity),
|
||||
.vendor = report.vendor,
|
||||
.device = report.device,
|
||||
.subsystem = report.subsystem,
|
||||
.hid = report.hid[0..hid_len],
|
||||
};
|
||||
}
|
||||
|
||||
@@ -365,6 +345,10 @@ fn sweepDeadlines() void {
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
manager_endpoint = endpoint;
|
||||
|
||||
// Load the authoritative driver-match registry before any bus driver can
|
||||
// report a device to match against it.
|
||||
loadRegistry();
|
||||
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = logging.write("/system/services/device-manager: out of memory\n");
|
||||
@@ -459,24 +443,22 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
if (status == 0) publishEvent(message[0..device_manager_protocol.child_added_size]);
|
||||
// Matching from reports (M19.3): a registered child whose identity
|
||||
// names a driver gets one, once — re-reports after a bus restart
|
||||
// dedupe on the registered id, exactly like the registrations do.
|
||||
// Matching from reports (M19.3), now data-driven via the /etc/devices.csv
|
||||
// registry: a registered child gets the most-specific driver its identity
|
||||
// matches, once — re-reports after a bus restart dedupe on the registered
|
||||
// id, exactly like the registrations do.
|
||||
if (status == 0 and report.device_id != device_manager_protocol.no_device) {
|
||||
if (pciDriverForIdentity(report.identity)) |child_driver| {
|
||||
if (!driverForDevice(report.device_id)) addDriver(child_driver, report.device_id, true);
|
||||
}
|
||||
// USB interface match: the reported identity is the packed class triple,
|
||||
// and the class driver is spawned with the interface's registered id.
|
||||
if (usbDriverForIdentity(report.identity)) |usb_driver| {
|
||||
if (!driverForDevice(report.device_id)) addDriver(usb_driver, report.device_id, true);
|
||||
}
|
||||
// ACPI _HID match (M20.3): ps2-bus is a singleton that finds its own
|
||||
// devices by hid, so spawn it once, without a device assignment.
|
||||
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
||||
if (hid_len != 0) {
|
||||
if (hidDriverFor(report.hid[0..hid_len])) |hid_driver| {
|
||||
if (!alreadySupervised(hid_driver)) addDriver(hid_driver, device_manager_protocol.no_device, false);
|
||||
const id = identityFromReport(report);
|
||||
if (registry.matchDriver(registry_rules[0..registry_count], id)) |match| {
|
||||
if (match.ambiguous)
|
||||
std.log.info("/etc/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
||||
if (id.bus == .acpi) {
|
||||
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
||||
// own devices once spawned — spawn it once, no device assignment.
|
||||
if (!alreadySupervised(match.driver)) addDriver(match.driver, device_manager_protocol.no_device, false);
|
||||
} else {
|
||||
// A per-device driver: one instance, the registered id as argv[1].
|
||||
if (!driverForDevice(report.device_id)) addDriver(match.driver, report.device_id, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -118,7 +118,17 @@ fn tryBringUp() void {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
};
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent) orelse return;
|
||||
// Shareable so the buffer's capability can be attached down the chain (block server
|
||||
// -> controller), making its physical addresses reachable by the device under an
|
||||
// enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n");
|
||||
return;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
|
||||
@@ -25,42 +25,88 @@ const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const power_protocol = @import("power-protocol");
|
||||
const build_options = @import("build_options");
|
||||
const fs = @import("file-system");
|
||||
const csv = @import("csv");
|
||||
|
||||
/// The system services init brings up at boot, in order, by binary path. This is
|
||||
/// init's policy — the microkernel keeps such choices in user space, not the
|
||||
/// kernel. Drivers are absent on purpose: the device manager owns those. (A
|
||||
/// future init reads this from a manifest under /system/services instead of a
|
||||
/// hardcoded list.)
|
||||
const boot_services = if (build_options.diagnose) [_][]const u8{
|
||||
// The diagnose boot: no display service, so the kernel's on-screen boot
|
||||
// transcript is never suppressed — the timestamped timeline (USB bring-up,
|
||||
// storage, logger) stays readable on real hardware with no serial.
|
||||
"/system/services/input",
|
||||
"/system/services/device-manager",
|
||||
"/system/services/fat",
|
||||
"/system/services/logger",
|
||||
} else [_][]const u8{
|
||||
"/system/services/input",
|
||||
"/system/services/device-manager",
|
||||
"/system/services/fat",
|
||||
"/system/services/display",
|
||||
"/system/services/display-demo",
|
||||
// Last: at shutdown children stop in reverse order, so the logger goes down
|
||||
// FIRST — its final drain still has the fat server (and the whole storage
|
||||
// chain) alive underneath it.
|
||||
"/system/services/logger",
|
||||
/// The system services init brings up at boot are init's policy, not the kernel's —
|
||||
/// and that policy is now data: `/etc/init.csv` (see `loadServices`), read at
|
||||
/// startup instead of a hardcoded list. Drivers are absent on purpose: the device
|
||||
/// manager owns those.
|
||||
///
|
||||
/// The most services `/etc/init.csv` can list, and the most argv entries (beyond the
|
||||
/// path) each may carry. Fixed caps because init parses the list into static storage —
|
||||
/// the freestanding, no-allocator counterpart to the device manager's registry table.
|
||||
const max_services = 16;
|
||||
const max_service_args = 4;
|
||||
|
||||
/// One service init starts, parsed from a row of `/etc/init.csv`: its binary path
|
||||
/// and argv, both slices into `init_csv` (held for the life of the process).
|
||||
const Service = struct {
|
||||
path: []const u8 = "",
|
||||
arg_buffer: [max_service_args][]const u8 = undefined,
|
||||
arg_count: usize = 0,
|
||||
fn arguments(self: *const Service) []const []const u8 {
|
||||
return self.arg_buffer[0..self.arg_count];
|
||||
}
|
||||
};
|
||||
|
||||
/// The live process id of each boot service (0 = not running), indexed by its position
|
||||
/// in `boot_services`, plus how many times init has restarted it. init supervises these:
|
||||
/// it spawns them against `supervision_endpoint` and, on a child's death, restarts it (up
|
||||
/// to `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md), the
|
||||
/// service-level counterpart to the device manager's driver restarts.
|
||||
var child_ids: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var restart_counts: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
/// The `/etc/init.csv` bytes, held because the parsed services slice into them.
|
||||
var init_csv: [4096]u8 = undefined;
|
||||
var services: [max_services]Service = .{Service{}} ** max_services;
|
||||
var service_count: usize = 0;
|
||||
|
||||
/// The live process id of each service (0 = not running) and its restart count,
|
||||
/// indexed by position in `services`. init supervises these: it spawns them against
|
||||
/// `supervision_endpoint` and, on a child's death, restarts it (up to
|
||||
/// `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md),
|
||||
/// the service-level counterpart to the device manager's driver restarts.
|
||||
var child_ids: [max_services]u32 = .{0} ** max_services;
|
||||
var restart_counts: [max_services]u32 = .{0} ** max_services;
|
||||
var shutting_down = false;
|
||||
var supervision_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// Parse `/etc/init.csv` into `services`, in file order (startup order; shutdown is
|
||||
/// the reverse). Each row is a binary path followed by its argv, comma-separated;
|
||||
/// `#` comments and blank lines are ignored. The file lives in the initial ramdisk,
|
||||
/// which the kernel serves directly, so init — PID 1, running before any filesystem
|
||||
/// service — reads it with a plain fs.open, the same mechanism the device manager
|
||||
/// uses for /etc/devices.csv. A missing file means no services (the no-ramdisk
|
||||
/// isolation test): loud, but not fatal.
|
||||
fn loadServices() void {
|
||||
var file = fs.open("/etc/init.csv", .{}) orelse {
|
||||
_ = logging.write("/system/services/init: /etc/init.csv missing — no services started\n");
|
||||
return;
|
||||
};
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < init_csv.len) {
|
||||
const n = file.read(init_csv[used..]) orelse break;
|
||||
if (n == 0) break;
|
||||
used += n;
|
||||
}
|
||||
var lines = std.mem.splitScalar(u8, init_csv[0..used], '\n');
|
||||
while (lines.next()) |line| {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) continue;
|
||||
if (service_count >= services.len) {
|
||||
_ = logging.write("/system/services/init: /etc/init.csv has more services than the table holds\n");
|
||||
break;
|
||||
}
|
||||
var it = csv.fields(body);
|
||||
const path = it.next() orelse continue;
|
||||
if (path.len == 0) continue;
|
||||
var service: Service = .{ .path = path };
|
||||
while (it.next()) |argument| {
|
||||
if (argument.len == 0) continue; // padding, or a trailing comma
|
||||
if (service.arg_count >= max_service_args) break;
|
||||
service.arg_buffer[service.arg_count] = argument;
|
||||
service.arg_count += 1;
|
||||
}
|
||||
services[service_count] = service;
|
||||
service_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||
/// that faults immediately on every spawn doesn't respawn forever.
|
||||
const maximum_restarts = 3;
|
||||
@@ -89,11 +135,12 @@ pub fn main() void {
|
||||
};
|
||||
_ = process.bindSignals(supervision_endpoint);
|
||||
|
||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||
// Best-effort and silent: each service announces its own readiness, and in
|
||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||
for (boot_services, 0..) |service, i| {
|
||||
if (process.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
|
||||
// Load the service list, then bring each up supervised so init can stop them
|
||||
// cleanly. Best-effort and silent: each service announces its own readiness,
|
||||
// and with no /etc/init.csv (an isolation test) the loop starts nothing.
|
||||
loadServices();
|
||||
for (services[0..service_count], 0..) |*service, i| {
|
||||
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
||||
}
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
@@ -141,22 +188,22 @@ pub fn main() void {
|
||||
/// iron rule 1); init only decides whether to bring it back.
|
||||
fn restartChild(id: u32) void {
|
||||
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||
for (boot_services, 0..) |service, i| {
|
||||
for (services[0..service_count], 0..) |*service, i| {
|
||||
if (child_ids[i] != id) continue;
|
||||
child_ids[i] = 0;
|
||||
// An unknown reason (the record aged out) is treated as a crash worth restarting.
|
||||
const reason = process.exitReason(id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
std.log.info("{s} exited cleanly; not restarting", .{service});
|
||||
std.log.info("{s} exited cleanly; not restarting", .{service.path});
|
||||
return;
|
||||
}
|
||||
restart_counts[i] += 1;
|
||||
if (restart_counts[i] > maximum_restarts) {
|
||||
std.log.info("{s} keeps crashing; giving up after {d} restarts", .{ service, maximum_restarts });
|
||||
std.log.info("{s} keeps crashing; giving up after {d} restarts", .{ service.path, maximum_restarts });
|
||||
return;
|
||||
}
|
||||
std.log.info("{s} died ({s}); restarting ({d}/{d})", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||
if (process.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||
std.log.info("{s} died ({s}); restarting ({d}/{d})", .{ service.path, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||
return;
|
||||
}
|
||||
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||
@@ -190,7 +237,7 @@ fn shutDown() void {
|
||||
// Log persistence is the logger service's job: it is the LAST boot service,
|
||||
// so the reverse-order stop below terminates it first and its final drain
|
||||
// runs while the whole storage chain is still alive.
|
||||
var i = boot_services.len;
|
||||
var i = service_count;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
if (child_ids[i] != 0) process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||
|
||||
+70
-8
@@ -146,20 +146,70 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# DMA memory (M14): contiguous frame allocation, below-4G cap, coherent mapping,
|
||||
# and reclaim on teardown.
|
||||
# Boots with no IOMMU device, so it also asserts the explicit fail-open boot line
|
||||
# (the DMA-isolation posture must be stated, never silent).
|
||||
{"name": "dma",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"expect": r"(?s)(?=.*iommu : none present - DMA fail-open \(unisolated\))(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# MSI (M15): allocate a per-device vector and deliver it as a notification (a
|
||||
# self-IPI stands in for the device's MSI write, since the HPET has no MSI).
|
||||
{"name": "msi",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IOMMU (M16): boot with an emulated VT-d unit and confirm danos parses the DMAR
|
||||
# table and reads the unit's registers. Detection only — enforcement is future.
|
||||
# IOMMU: boot with an emulated VT-d unit; danos parses the DMAR table, enables
|
||||
# translation, and proves the domain walker (map/resolve/unmap on a scratch domain)
|
||||
# with no spurious faults. The `enabled` line is a lookahead so a silently-dead unit
|
||||
# cannot fake a pass.
|
||||
{"name": "iommu",
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
"expect": r"(?s)(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA under translation: the full USB storage stack (xHC ring DMA + BOT/SCSI + fat's
|
||||
# cross-process bounce buffer) runs with VT-d enabled. Every device is identity-
|
||||
# mapped in the blanket domain, so DMA works, but through real second-level walks.
|
||||
{"name": "iommu-usb-storage",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||
# level translation with interrupt remapping off).
|
||||
{"name": "iommu-usb-hid",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Enforcement, the negative proof: a claimed e1000e fires a DMA at an unmapped page;
|
||||
# VT-d must fault it (logged) and the system must stay alive. The fault line is the
|
||||
# point here, so unlike the positive cases it appears in `expect`, not `fail`.
|
||||
{"name": "iommu-fault",
|
||||
"smp": 4,
|
||||
"timeout": 120,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off", "-device", "e1000e"],
|
||||
"expect": r"(?s)(?=.*DANOS-IOMMU-FAULT: bdf=)(?=.*iommu-fault-test: system alive)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"iommu-fault-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# AMD-Vi: the same detection + scratch-domain walker proof as the `iommu` case, but on
|
||||
# the AMD backend (IVRS parse, device table, command buffer). QEMU's amd-iommu needs
|
||||
# dma-remap=on (default off = translation silently ignored). UNTESTED on real AMD.
|
||||
{"name": "amd-iommu",
|
||||
"build_case": "iommu",
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA under AMD-Vi translation: the full USB storage stack through AMD device-table
|
||||
# translation. Tentative — land depends on QEMU amd-iommu behaving. UNTESTED on real AMD.
|
||||
{"name": "amd-iommu-usb-storage",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||
{"name": "ioport",
|
||||
@@ -628,7 +678,7 @@ CASES = [
|
||||
{"name": "acpi-ps2",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"discovery: reported PNP0303[\s\S]*"
|
||||
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[\s\S]*"
|
||||
r"device-manager: spawned \S*ps2-bus[\s\S]*"
|
||||
r"ps2-bus: keyboard driver attached",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -675,8 +725,8 @@ CASES = [
|
||||
{"name": "acpi-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"discovery: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||
r"discovery: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[^\n]*\(3 resources\)[\s\S]*"
|
||||
r"discovery: device \d+\s+bus=acpi hid=PNP0F13[^\n]*\(1 resources\)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M19.1/M19.3: the ring-3 PCI scan. pci-bus walks the ECAM through its mmio_map
|
||||
# grant and registers every function it finds; the kernel's own walk retired, so
|
||||
@@ -687,6 +737,18 @@ CASES = [
|
||||
# duplicates); this ordered regex asserts the drill itself over the whole serial
|
||||
# log — the backreference requires the respawn to re-scan the same count, and the
|
||||
# full-capture match is immune to the transient-line races an in-kernel poll hits.
|
||||
# The driver-side PCI library (library/device/pci) against a real function: an extra
|
||||
# e1000e NIC — PM + MSI + PCIe + MSI-X capabilities, claimed by no danos driver — is
|
||||
# claimed by the pci-cap-test fixture, which exercises the capability walks, MSI
|
||||
# programming (the first driver-side msi_bind use), the MSI-X table, power state,
|
||||
# and FLR, printing a marker per check.
|
||||
{"name": "pci-caps",
|
||||
"smp": 4,
|
||||
"timeout": 120,
|
||||
"qemu_extra": ["-device", "e1000e"],
|
||||
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*pci-cap-test: all checks passed)",
|
||||
"fail": r"pci-cap-test: FAIL|DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
{"name": "pci-scan",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
//! iommu-fault-test — the negative proof for IOMMU enforcement. The iommu-fault QEMU
|
||||
//! case boots with VT-d enabled and an extra e1000e NIC no danos driver claims; this
|
||||
//! fixture claims it, then deliberately programs its transmit engine to DMA from a
|
||||
//! physical address that was never dma_alloc'd (so it is in no device's domain). The
|
||||
//! IOMMU must fault that access — the descriptor fetch never reaches memory — and the
|
||||
//! system must stay alive. A kernel `DANOS-IOMMU-FAULT` line plus this fixture's
|
||||
//! `system alive` marker is the pass.
|
||||
//!
|
||||
//! The rogue target is the e1000e's transmit descriptor RING base itself: the very first
|
||||
//! DMA the engine issues on a doorbell write is the descriptor fetch from that base, so
|
||||
//! pointing the ring at an unmapped page makes the first access the faulting one — no
|
||||
//! valid descriptor need be crafted.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
const mmio = @import("mmio");
|
||||
const pci = @import("pci");
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
const intel_vendor: u16 = 0x8086;
|
||||
const e1000e_device: u16 = 0x10D3;
|
||||
|
||||
const ethernet_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.network),
|
||||
.subclass = @intFromEnum(pci_class.network.SubClass.ethernet),
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
// e1000e transmit-engine registers (Intel 82574L datasheet §Register Descriptions),
|
||||
// byte offsets within BAR0. VERIFY-AGAINST-SPEC held on first bring-up: these are the
|
||||
// legacy TX ring registers.
|
||||
const reg_tctl = 0x0400; // Transmit Control
|
||||
const reg_tdbal = 0x3800; // TX Descriptor Base Address Low
|
||||
const reg_tdbah = 0x3804; // TX Descriptor Base Address High
|
||||
const reg_tdlen = 0x3808; // TX Descriptor Length (bytes, 128-byte aligned)
|
||||
const reg_tdh = 0x3810; // TX Descriptor Head
|
||||
const reg_tdt = 0x3818; // TX Descriptor Tail
|
||||
|
||||
const tctl_en: u32 = 1 << 1; // Transmit Enable
|
||||
const tctl_psp: u32 = 1 << 3; // Pad Short Packets
|
||||
|
||||
/// A low physical page that user DMA never touches — never returned by dma_alloc (whose
|
||||
/// arena is far higher), so it is in no device's IOMMU domain. The e1000e's descriptor
|
||||
/// fetch from here is exactly the out-of-domain access the unit must block.
|
||||
const rogue_physical: u64 = 0x1000;
|
||||
|
||||
var descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const nic_id: u64 = found: {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 150) : (tries += 1) {
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
for (descriptors[0..@min(total, descriptors.len)]) |*entry| {
|
||||
if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) {
|
||||
descriptor = entry.*;
|
||||
break :found entry.id;
|
||||
}
|
||||
}
|
||||
time.sleepMillis(100);
|
||||
}
|
||||
_ = logging.write("iommu-fault-test: FAIL no ethernet function found\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(nic_id)) {
|
||||
_ = logging.write("iommu-fault-test: FAIL claim\n");
|
||||
return;
|
||||
}
|
||||
var function = pci.Function.map(nic_id, &descriptor) orelse {
|
||||
_ = logging.write("iommu-fault-test: FAIL config-space map\n");
|
||||
return;
|
||||
};
|
||||
if (function.vendorId() != intel_vendor or function.deviceId() != e1000e_device) {
|
||||
_ = logging.write("iommu-fault-test: FAIL not an e1000e\n");
|
||||
return;
|
||||
}
|
||||
function.enableMemoryAndBusMaster();
|
||||
const bar0 = function.mapBar(0) orelse {
|
||||
_ = logging.write("iommu-fault-test: FAIL map BAR0\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// Point the TX ring at the rogue page and kick the engine: TDBA = rogue, a non-zero
|
||||
// length, head=0, enable, then tail=1 so the engine fetches descriptor 0 — a DMA
|
||||
// read from the rogue page, which the IOMMU must fault.
|
||||
_ = logging.write("iommu-fault-test: pointing e1000e TX ring at an unmapped page\n");
|
||||
mmio.writeRegister(u32, bar0 + reg_tdbal, @truncate(rogue_physical));
|
||||
mmio.writeRegister(u32, bar0 + reg_tdbah, @intCast(rogue_physical >> 32));
|
||||
mmio.writeRegister(u32, bar0 + reg_tdlen, 128);
|
||||
mmio.writeRegister(u32, bar0 + reg_tdh, 0);
|
||||
mmio.writeRegister(u32, bar0 + reg_tctl, tctl_en | tctl_psp);
|
||||
mmio.writeMemoryBarrier();
|
||||
mmio.writeRegister(u32, bar0 + reg_tdt, 1); // doorbell: fetch descriptor 0
|
||||
|
||||
// Give the engine time to attempt the fetch, then force the fault records to the log.
|
||||
time.sleepMillis(200);
|
||||
const faults = device.iommuFaultDrain();
|
||||
writeLine("iommu-fault-test: drained {d} iommu fault(s)\n", .{faults});
|
||||
if (faults == 0) {
|
||||
_ = logging.write("iommu-fault-test: FAIL rogue DMA was not blocked\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Liveness: the system survived the blocked DMA — read our own config space back.
|
||||
if (function.vendorId() != intel_vendor) {
|
||||
_ = logging.write("iommu-fault-test: FAIL device unreadable after fault\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("iommu-fault-test: system alive\n");
|
||||
}
|
||||
@@ -0,0 +1,178 @@
|
||||
//! pci-cap-test — QEMU fixture for the driver-side PCI library (library/device/pci).
|
||||
//! The pci-caps test case boots with an extra `-device e1000e` NIC that no danos driver
|
||||
//! claims; this fixture claims it and exercises the whole claimed-function surface
|
||||
//! against real (emulated) hardware: header accessors, command bits, the capability
|
||||
//! walk, MSI programming (the first driver-side `msi_bind` use), the MSI-X table,
|
||||
//! extended capabilities, power state, and — where offered — function-level reset.
|
||||
//! Every check prints `pci-cap-test: <check> ok` or `pci-cap-test: FAIL <check>`; the
|
||||
//! harness asserts on the final `all checks passed` marker (test/qemu_test.py).
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
const mmio = @import("mmio");
|
||||
const pci = @import("pci");
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
/// QEMU's e1000e: Intel 82574L.
|
||||
const intel_vendor: u16 = 0x8086;
|
||||
const e1000e_device: u16 = 0x10D3;
|
||||
|
||||
const ethernet_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.network),
|
||||
.subclass = @intFromEnum(pci_class.network.SubClass.ethernet),
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
/// `pci.Function` keeps a pointer to the descriptor, so it must outlive the stack frame
|
||||
/// that found it.
|
||||
var descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Print the check's verdict; the caller returns on false to stop at the first failure.
|
||||
fn check(comptime name: []const u8, ok: bool) bool {
|
||||
if (ok) {
|
||||
_ = logging.write("pci-cap-test: " ++ name ++ " ok\n");
|
||||
} else {
|
||||
_ = logging.write("pci-cap-test: FAIL " ++ name ++ "\n");
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// The bus scan runs in another process; poll until the NIC shows up.
|
||||
const nic_id: u64 = found: {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 150) : (tries += 1) {
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
for (descriptors[0..@min(total, descriptors.len)]) |*entry| {
|
||||
if (entry.class == @intFromEnum(device.DeviceClass.pci_device) and entry.pci_class == ethernet_class) {
|
||||
descriptor = entry.*;
|
||||
break :found entry.id;
|
||||
}
|
||||
}
|
||||
time.sleepMillis(100);
|
||||
}
|
||||
_ = logging.write("pci-cap-test: FAIL no ethernet function found\n");
|
||||
return;
|
||||
};
|
||||
writeLine("pci-cap-test: claiming ethernet function (device {d})\n", .{nic_id});
|
||||
if (!check("claim", device.claim(nic_id))) return;
|
||||
var function = pci.Function.map(nic_id, &descriptor) orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL config-space map\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// Identity: the header accessors against known e1000e values.
|
||||
if (!check("vendor/device id", function.vendorId() == intel_vendor and function.deviceId() == e1000e_device)) return;
|
||||
if (!check("class code", function.classCode().pack() == ethernet_class)) return;
|
||||
if (!check("subsystem ids readable", function.subsystemVendorId() != 0xFFFF and function.subsystemId() != 0xFFFF)) return;
|
||||
|
||||
// Command bits: enable, read back, quiesce, read back, re-enable.
|
||||
function.enableMemoryAndBusMaster();
|
||||
if (!check("memory+bus-master enable", function.command() & pci_class.command_memory_and_bus_master == pci_class.command_memory_and_bus_master)) return;
|
||||
function.disableBusMaster();
|
||||
if (!check("bus-master disable", function.command() & pci_class.command_bus_master == 0)) return;
|
||||
function.enableMemoryAndBusMaster();
|
||||
|
||||
// The capability walk: e1000e advertises PM, MSI, PCIe, and MSI-X.
|
||||
var seen_power = false;
|
||||
var seen_msi = false;
|
||||
var seen_pci_express = false;
|
||||
var seen_msix = false;
|
||||
var walk = function.capabilities();
|
||||
while (walk.next()) |capability| {
|
||||
switch (capability.id) {
|
||||
@intFromEnum(pci_class.CapabilityId.power_management) => seen_power = true,
|
||||
@intFromEnum(pci_class.CapabilityId.msi) => seen_msi = true,
|
||||
@intFromEnum(pci_class.CapabilityId.pci_express) => seen_pci_express = true,
|
||||
@intFromEnum(pci_class.CapabilityId.msix) => seen_msix = true,
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
if (!check("capability walk", seen_power and seen_msi and seen_pci_express and seen_msix)) return;
|
||||
if (!check("findCapability", function.findCapability(.msi) != null and function.findCapability(.pci_express) != null)) return;
|
||||
|
||||
// MSI: bind a vector (the syscall's first driver-side use), program the capability,
|
||||
// and read the registers straight back.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL endpoint creation\n");
|
||||
return;
|
||||
};
|
||||
const message = device.msiBind(nic_id, endpoint) orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL msi_bind\n");
|
||||
return;
|
||||
};
|
||||
if (!check("msi_bind address", message.address == 0xFEE0_0000)) return;
|
||||
if (!check("programMsi", function.programMsi(message))) return;
|
||||
const msi_cap = function.findCapability(.msi).?;
|
||||
const msi_control = mmio.readRegister(u16, msi_cap.offset + pci_class.msi.control);
|
||||
const msi_data_offset: usize = if (msi_control & pci_class.msi.control_64bit_capable != 0) pci_class.msi.data_64 else pci_class.msi.data_32;
|
||||
if (!check("msi registers read back", msi_control & pci_class.msi.control_enable != 0 and
|
||||
msi_control & pci_class.msi.control_multiple_message_enable_mask == 0 and
|
||||
mmio.readRegister(u32, msi_cap.offset + pci_class.msi.address) == @as(u32, @truncate(message.address)) and
|
||||
mmio.readRegister(u16, msi_cap.offset + msi_data_offset) == @as(u16, @truncate(message.data)))) return;
|
||||
if (!check("intx disabled with msi", function.command() & pci_class.command_interrupt_disable != 0)) return;
|
||||
function.disableMsi();
|
||||
if (!check("disableMsi", mmio.readRegister(u16, msi_cap.offset + pci_class.msi.control) & pci_class.msi.control_enable == 0)) return;
|
||||
|
||||
// MSI-X: map the table, program entry 0, exercise the masks. Never enable — this
|
||||
// proves the programming surface, not delivery.
|
||||
const msix_table = function.msix() orelse {
|
||||
_ = logging.write("pci-cap-test: FAIL msix table map\n");
|
||||
return;
|
||||
};
|
||||
writeLine("pci-cap-test: msix table has {d} entries\n", .{msix_table.entry_count});
|
||||
if (!check("msix entry count", msix_table.entry_count >= 1)) return;
|
||||
if (!check("msix programEntry", msix_table.programEntry(0, message))) return;
|
||||
if (!check("msix entry reads back", mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_address) == @as(u32, @truncate(message.address)) and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_data) == message.data and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked != 0)) return;
|
||||
if (!check("msix unmask entry", msix_table.unmaskEntry(0) and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked == 0)) return;
|
||||
if (!check("msix re-mask entry", msix_table.maskEntry(0) and
|
||||
mmio.readRegister(u32, msix_table.table + pci_class.msix.entry_vector_control) & pci_class.msix.entry_vector_control_masked != 0)) return;
|
||||
msix_table.setFunctionMask();
|
||||
if (!check("msix function mask", mmio.readRegister(u16, msix_table.capability + pci_class.msix.control) & pci_class.msix.control_function_mask != 0)) return;
|
||||
msix_table.clearFunctionMask();
|
||||
if (!check("msix function unmask", mmio.readRegister(u16, msix_table.capability + pci_class.msix.control) & pci_class.msix.control_function_mask == 0)) return;
|
||||
if (!check("msix out-of-range rejected", !msix_table.programEntry(msix_table.entry_count, message))) return;
|
||||
|
||||
// Extended capabilities: the walk must terminate cleanly; the count is informative
|
||||
// (don't hard-bind to QEMU's exact extended-capability set).
|
||||
var extended_count: u32 = 0;
|
||||
var extended = function.extendedCapabilities();
|
||||
while (extended.next()) |_| extended_count += 1;
|
||||
writeLine("pci-cap-test: {d} extended capabilities\n", .{extended_count});
|
||||
if (!check("extended walk terminates", extended_count < 480)) return;
|
||||
|
||||
// Power: QEMU leaves the function in D0; ensurePowerStateD0 must agree and not
|
||||
// disturb the PMCSR.
|
||||
const power_cap = function.findCapability(.power_management).?;
|
||||
const pmcsr_before = mmio.readRegister(u16, power_cap.offset + pci_class.power_management.control_status);
|
||||
if (!check("power state is D0", pmcsr_before & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0)) return;
|
||||
function.ensurePowerStateD0();
|
||||
if (!check("ensurePowerStateD0 is a no-op at D0", mmio.readRegister(u16, power_cap.offset + pci_class.power_management.control_status) == pmcsr_before)) return;
|
||||
|
||||
// Function-level reset, where the device offers it: afterwards the function must be
|
||||
// readable with its identity intact, and bring-up must work again.
|
||||
const express_cap = function.findCapability(.pci_express).?;
|
||||
const device_capabilities = mmio.readRegister(u32, express_cap.offset + pci_class.pci_express.device_capabilities);
|
||||
if (device_capabilities & pci_class.pci_express.device_capabilities_flr != 0) {
|
||||
if (!check("functionLevelReset", function.functionLevelReset())) return;
|
||||
if (!check("identity after flr", function.vendorId() == intel_vendor and function.deviceId() == e1000e_device)) return;
|
||||
function.enableMemoryAndBusMaster();
|
||||
if (!check("re-enable after flr", function.command() & pci_class.command_memory_and_bus_master == pci_class.command_memory_and_bus_master)) return;
|
||||
} else {
|
||||
_ = logging.write("pci-cap-test: flr not offered, skipped\n");
|
||||
}
|
||||
|
||||
_ = logging.write("pci-cap-test: all checks passed\n");
|
||||
}
|
||||
Reference in New Issue
Block a user