diff --git a/build.zig b/build.zig index e955750..6fdb186 100644 --- a/build.zig +++ b/build.zig @@ -51,13 +51,13 @@ fn timestamp(b: *std.Build) []const u8 { /// Build one user-space binary the same way for every program (init, and later /// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model /// (the image base is above 4 GiB — smaller models emit 32-bit relocations that -/// can't reach), linked against the `rt` runtime library with the shared user +/// can't reach), linked against the `runtime` runtime library with the shared user /// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions) /// are authoritative — the kernel's W^X user-ELF loader requires exact perms. fn addUserBinary( b: *std.Build, target: std.Build.ResolvedTarget, - rt_mod: *std.Build.Module, + runtime_module: *std.Build.Module, name: []const u8, root: []const u8, ) *std.Build.Step.Compile { @@ -73,7 +73,7 @@ fn addUserBinary( .stack_check = false, .stack_protector = false, .imports = &.{ - .{ .name = "rt", .module = rt_mod }, + .{ .name = "runtime", .module = runtime_module }, }, }), }); @@ -91,74 +91,74 @@ pub fn build(b: *std.Build) void { const target = b.standardTargetOptions(.{}); const optimize = b.standardOptimizeOption(.{}); - // Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set, + // Shared handoff definitions (BootInformation, Framebuffer, ...). No target is set, // so the module inherits the target of whichever binary imports it — the // freestanding kernel or the UEFI bootloader. - const mod = b.addModule("danos", .{ + const danos_module = b.addModule("danos", .{ .root_source_file = b.path("src/root.zig"), }); - // Kernel tunables (max_cpus, stack sizes, tick rate). A dependency-free module of + // Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of // compile-time constants, imported wherever a knob is read; keeps the trade-offs - // in one place instead of scattered across the tree. See src/config.zig. - const config_mod = b.addModule("config", .{ - .root_source_file = b.path("src/config.zig"), + // in one place instead of scattered across the tree. See src/configuration.zig. + const parameters_module = b.addModule("parameters", .{ + .root_source_file = b.path("src/parameters.zig"), }); // Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging). - // The generic kernel imports this as "arch" and never names x86_64, so a new + // The generic kernel imports this as "architecture" and never names x86_64, so a new // architecture is a matter of pointing this module at a different directory. - const arch_mod = b.addModule("arch", .{ + const architecture_module = b.addModule("architecture", .{ .root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"), .imports = &.{ - .{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types - .{ .name = "config", .module = config_mod }, // max_cpus, ist_stack_size, timer_hz + .{ .name = "danos", .module = danos_module }, // paging uses the shared BootInformation/memory-map types + .{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz }, }); // CPU-exception stubs — real assembly, since they need cross-symbol // jumps/calls that Zig inline asm can't express (see the file's header). - arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s")); + architecture_module.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s")); // The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be // inline asm (it runs relocated to a low page, not at its link address). - arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s")); + architecture_module.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s")); // Firmware-agnostic device discovery. The generic kernel imports this as // "platform" and asks it to enumerate hardware into a backend-neutral device // tree, never naming ACPI (or, later, device-tree) — the same discipline the - // arch module applies to CPU code. The backend is selected at runtime from + // architecture module applies to CPU code. The backend is selected at runtime from // the boot handoff (see src/device/platform.zig). - const platform_mod = b.addModule("platform", .{ + const platform_module = b.addModule("platform", .{ .root_source_file = b.path("src/device/platform.zig"), .imports = &.{ - .{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP) - .{ .name = "config", .module = config_mod }, // max_cpus (the discovery pool) + .{ .name = "danos", .module = danos_module }, // BootInformation (carries the ACPI RSDP) + .{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool) }, }); - // The user-space runtime library (a nascent libc): syscall wrappers, the + // The user-space runtime library (a nascent libc): system_call wrappers, the // C-convention heap, IPC helpers, the process start shim. Compiled into every // user binary (see addUserBinary), so it inherits each exe's `.large` code // model — do NOT set a target/code_model here. It imports `danos` for the - // shared Syscall numbers. - const rt_mod = b.addModule("rt", .{ - .root_source_file = b.path("lib/rt.zig"), + // shared SystemCall numbers. + const runtime_module = b.addModule("runtime", .{ + .root_source_file = b.path("lib/runtime.zig"), .imports = &.{ - .{ .name = "danos", .module = mod }, + .{ .name = "danos", .module = danos_module }, }, }); // The initrd container format, shared by the kernel (unpacks it) and the // build-time packer tools/mkinitrd.zig (produces it). No dependencies. - const initrd_mod = b.addModule("initrd", .{ - .root_source_file = b.path("src/user/proto/initrd.zig"), + const initrd_module = b.addModule("initrd", .{ + .root_source_file = b.path("src/user/protocol/initrd.zig"), }); - // Compile-time config the kernel reads as `@import("build_options")`. The + // Compile-time configuration the kernel reads as `@import("build_options")`. The // QEMU test harness sets -Dtest-case= to run one self-test at boot. const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)"); const build_options = b.addOptions(); build_options.addOption(?[]const u8, "test_case", test_case); - const build_options_mod = build_options.createModule(); + const build_options_module = build_options.createModule(); // --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader --- // SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff, @@ -177,18 +177,18 @@ pub fn build(b: *std.Build) void { .target = kernel_target, .optimize = optimize, .code_model = .kernel, // kernel runs in the top 2 GiB (higher half) - .red_zone = false, // interrupts would corrupt the SysV red zone + .red_zone = false, // interrupts would corrupt the SystemV red zone .single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores .sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide .stack_check = false, // stack-probe calls have no runtime to land in .stack_protector = false, .imports = &.{ - .{ .name = "danos", .module = mod }, - .{ .name = "arch", .module = arch_mod }, - .{ .name = "platform", .module = platform_mod }, - .{ .name = "config", .module = config_mod }, - .{ .name = "build_options", .module = build_options_mod }, - .{ .name = "initrd", .module = initrd_mod }, + .{ .name = "danos", .module = danos_module }, + .{ .name = "architecture", .module = architecture_module }, + .{ .name = "platform", .module = platform_module }, + .{ .name = "parameters", .module = parameters_module }, + .{ .name = "build_options", .module = build_options_module }, + .{ .name = "initrd", .module = initrd_module }, }, }), }); @@ -208,18 +208,19 @@ pub fn build(b: *std.Build) void { // --- /sbin/init: the first user-space program --- // Built by the shared user-binary recipe (see addUserBinary): freestanding, - // linked into the kernel's user region against the `rt` runtime library, and + // linked into the kernel's user region against the `runtime` runtime library, and // started in ring 3 by the kernel's user-ELF loader. - const init_exe = addUserBinary(b, kernel_target, rt_mod, "init", "sbin/init.zig"); + const init_exe = addUserBinary(b, kernel_target, runtime_module, "init", "sbin/init.zig"); b.installArtifact(init_exe); // --- initrd: a bundle of extra user binaries (VFS server + drivers) --- // Each is built by the same user-binary recipe, then packed into one image by // the host-side mkinitrd tool. The bootloader ferries the image to the kernel, - // which unpacks it and spawns each program (src/user/proto/initrd.zig). - const vfs_exe = addUserBinary(b, kernel_target, rt_mod, "vfs", "sbin/vfs.zig"); - const vfstest_exe = addUserBinary(b, kernel_target, rt_mod, "vfstest", "sbin/vfstest.zig"); - const hpetd_exe = addUserBinary(b, kernel_target, rt_mod, "hpetd", "sbin/hpetd.zig"); + // which unpacks it and spawns each program (src/user/protocol/initrd.zig). + const vfs_exe = addUserBinary(b, kernel_target, runtime_module, "vfs", "sbin/vfs.zig"); + const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, "vfs-test", "sbin/vfs-test.zig"); + const hpetd_exe = addUserBinary(b, kernel_target, runtime_module, "hpetd", "sbin/hpetd.zig"); + const busd_exe = addUserBinary(b, kernel_target, runtime_module, "busd", "sbin/busd.zig"); // Pack the user binaries into the initrd image with the host-side Python tool // (the container format is trivial, and Python sidesteps std API churn). Args: @@ -229,10 +230,12 @@ pub fn build(b: *std.Build) void { const initrd_img = mk_run.addOutputFileArg("initrd.img"); mk_run.addArg("vfs"); mk_run.addFileArg(vfs_exe.getEmittedBin()); - mk_run.addArg("vfstest"); + mk_run.addArg("vfs-test"); mk_run.addFileArg(vfstest_exe.getEmittedBin()); mk_run.addArg("hpetd"); mk_run.addFileArg(hpetd_exe.getEmittedBin()); + mk_run.addArg("busd"); + mk_run.addFileArg(busd_exe.getEmittedBin()); // Install the image to zig-out/bin (so the QEMU test harness picks it up like // the other binaries). The run-x86-64 ESP install is added below. @@ -252,7 +255,7 @@ pub fn build(b: *std.Build) void { }), .optimize = optimize, .imports = &.{ - .{ .name = "danos", .module = mod }, + .{ .name = "danos", .module = danos_module }, }, }), }); @@ -261,14 +264,14 @@ pub fn build(b: *std.Build) void { // --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF --- // Firmware lives in different places per OS/distro, so probe the known - // layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first + // layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first // that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere. const ovmf_code = b.option( []const u8, "ovmf-code", "Path to the OVMF_CODE firmware image", ) orelse firstExisting(b.graph.io, &.{ - "/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch + "/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture "/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu "/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu "/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora @@ -280,7 +283,7 @@ pub fn build(b: *std.Build) void { "ovmf-vars", "Path to the OVMF_VARS firmware image (a writable copy is made)", ) orelse firstExisting(b.graph.io, &.{ - "/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch + "/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture "/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu "/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu "/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora diff --git a/docs/README.md b/docs/README.md index 3cb78d1..ebd66ca 100644 --- a/docs/README.md +++ b/docs/README.md @@ -35,9 +35,20 @@ rather than restate it. Roughly in the order things happen at runtime: multitasking: kernel threads, the context switch, O(1) priority selection, and blocking (sleep, wait queues) — the leap to a running system. 11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking - message-passing channels — the backbone the microkernel's isolated servers will - talk over. -12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and + message-passing channels, then synchronous call/reply between *processes* over + endpoints — the backbone the microkernel's isolated servers talk over. +12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for + something: the `syscall`/`sysret` fast path, the trap frame, and why the table is + deliberately tiny. +13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an + ordinary ring-3 process that claims a device, maps its registers, and **sleeps + until its hardware interrupts it**. The claim is the capability; `irq_ack` is the + unmask. +14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How + real driver stacks factor into three shapes, how families share code, and the + proposed ABI for the three primitives still missing (capability passing, DMA + + memory barriers, MSI). +15. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and how `while (true) hlt` parks the CPU safely once there's nothing left to do. Start with the north star: @@ -68,6 +79,10 @@ Cutting across all of these: - **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels (L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the right choice depends on whether danos is chasing real-time or resilience. +- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the + tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are + `kebab-case`, code follows Zig's case conventions, and the handful of exceptions + (POSIX/C ABI names, `init`/`len`/`ptr`, acronyms). - **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means, and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff). - **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in @@ -94,19 +109,34 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts** ([halting.md](halting.md)). +Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md), +[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for +things through the small **[syscall](syscall.md)** table, isolated servers reach each +other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims +a device, maps its registers, and sleeps until the hardware interrupts it — which is +the whole reason for the arrangement ([vision.md](vision.md)). + ## Source map | Area | Code | |------|------| | Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` | | Kernel entry, panic, bring-up | `src/kernel/main.zig` | -| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` | +| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, `Syscall`, ABI) | `src/root.zig` | | Physical frame allocator | `src/kernel/pmm.zig` | | Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` | -| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` | -| IPC channels (message passing) | `src/kernel/ipc.zig` | +| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/scheduler.zig` | +| Big kernel lock + interrupt-safe critical sections | `src/kernel/sync.zig` | +| IPC channels between kernel threads (message passing) | `src/kernel/ipc.zig` | +| IPC endpoints: cross-address-space call/reply, handles, notifications | `src/kernel/ipc-synchronous.zig` | +| User processes: ELF loading, address spaces, the syscall table | `src/kernel/process.zig` | +| Device tree + claim capability + `device_register` containment | `src/kernel/device-service.zig` | +| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `src/kernel/irq.zig` | +| Hardware discovery (ACPI/device tree) behind one neutral device model | `src/device/` | | Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` | | In-kernel test cases | `src/kernel/tests.zig` | -| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` | +| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` | +| User runtime library (`rt`): syscalls, heap, stdio, IPC, device access | `lib/` | +| User-space programs shipped in the initrd (`init`, `vfs`, `hpetd` leaf driver, `busd` bus driver) | `sbin/` | | Build + `run-x86-64` (QEMU/OVMF) | `build.zig` | | QEMU integration test harness | `test/qemu_test.py` | diff --git a/docs/coding-standards.md b/docs/coding-standards.md new file mode 100644 index 0000000..3edbfe5 --- /dev/null +++ b/docs/coding-standards.md @@ -0,0 +1,137 @@ +# Coding standards + +Conventions for danos source. The overriding one, from which most of the rest follows: + +> **Names are spelled out in full. An identifier is not abbreviated unless the +> abbreviation is an acronym.** + +`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len` +is a Zig idiom — see the exceptions). `device_service`, not `device_service`. `scheduler`, not +`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a +cryptic one is paid every time the code is read, by everyone who reads it. In a +microkernel whose whole argument is that a human can hold each piece in their head, +that trade is not close. + +## The rule, precisely + +**Acronyms and initialisms stay.** They *are* the full name — expanding them would make +the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`, +`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`, +`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`, +`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`, +`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention +demands: `Hal` the type, `hal` the variable, `mapMmio` the function. + +**Everything else is spelled out.** If it's a word with letters removed, restore them: + +| Abbreviation | Full | +|---|---| +| `proto` | `protocol` | +| `msg` | `message` | +| `desc` | `descriptor` | +| `res` | `resource` | +| `recv` | `receive` | +| `buf` | `buffer` | +| `cur` | `current` | +| `src` / `dst` | `source` / `destination` | +| `idx` | `index` | +| `addr` | `address` | +| `reg` | `register` | +| `prev` | `previous` | +| `cfg` / `config` | `configuration` | +| `arch` | `architecture` | +| `sched` | `scheduler` | +| `dev` | `device` | +| `sys` / `syscall` | `system` / `system_call` | +| `info` | `information` | +| `dt` | `device_tree` | +| `ep` | `endpoint` | +| `rt` | `runtime` | +| `func` | `function` | +| `phys` / `virt` | `physical` / `virtual` | +| `wq` | `wait_queue` | + +This list is illustrative, not exhaustive. The rule is the rule; when you meet a new +abbreviation, expand it. + +## Exceptions + +Three, and only three. + +1. **Foreign ABI names are spelled exactly as the ABI spells them.** A function that + *is* the C or POSIX interface keeps its name: `fopen`, `fwrite`, `fread`, `malloc`, + `calloc`, `realloc`, `free`, `memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, + `close`, `lseek`, `stat`, `errno`. We don't get to rename `fwrite` to + `fileWrite` — it wouldn't be `fwrite` any more. This also covers the syscall + *wrappers* that exist to match those names. It does **not** license inventing new + abbreviated names in that style. + +2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's, + not ours, and are left alone: + - **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a + shortening of "initialize". + - **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own + structs use bare `len`/`ptr` fields to mirror them, so a reader carries one + mental model. (Compounds still expand: a field is `message_len`, not + `message_length` — `len` is kept, `msg` is not.) + - The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are + Zig's spelling. + + The rule governs the names *we* coin. + +3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may + keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the + letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it. + +4. **Established Unix filesystem and program conventions.** Top-level directories keep + their conventional names — `src`, `lib`, `sbin`, `bin`, `docs` — as do daemon + programs by their `d` suffix (`hpetd`, `busd`, following `sshd`/`httpd`). These are + names a Unix reader already knows; expanding them fights the convention rather than + serving it. + +## A note on collisions + +Two identifiers can legitimately expand to the same word. When they do, keep both +meaningful by renaming one to its *specific* identity rather than the generic +expansion. Two cases resolved this way: + +- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would + collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module + became **`parameters`**, which is what it holds. +- The kernel `device.zig` module would collide with `dev` (a device value) at + `device`. The module alias became **`device_model`**, which is what it is — the + device data model (`Device`, `DeviceTree`, `ResourceKind`). +- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a + `Namespace` **instance**. Resolved by dropping the module alias entirely — the two + types it provided are imported directly (`const Node = @import("namespace.zig").Node;`) + — which frees `namespace` for the instance. + +A related case is one abbreviation with two meanings. In the AML code, `op` means +**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` / +`LogicOperation` means **operation** — distinguished by case. The per-opcode parser +handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the +opcode's structure, which says what they do without overloading "op". + +## Case and file names + +Within those spelling rules, follow Zig's own conventions: + +- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`. +- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`. +- **Variables, fields, constants** — `snake_case`: `message_length`, `device_service`, + `notify_badge_bit`. + +**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it: +`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `device-service.zig`. A +single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`, +`idt.zig`. (The module *alias* a file is imported under still follows the code +conventions above — `snake_case` — because it's an identifier, not a filename.) + +## Why acronyms are the line + +Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped +input output" in code — that expansion is what the acronym *is for*. But `msg` is just +`message` with three letters stolen, and stealing them buys nothing a reader wants. The +test for "is this an abbreviation I must expand" is simply: *is there a longer word this +is a clipped form of?* If yes, write the word. If it's an initialism standing in for a +phrase, leave it. diff --git a/docs/device-interrupts.md b/docs/device-interrupts.md index 048e9ad..7c62ec3 100644 --- a/docs/device-interrupts.md +++ b/docs/device-interrupts.md @@ -89,7 +89,6 @@ if (state.vector < 32) { on_fault(state); // exception: report and halt (never returns) } else if (handlers[state.vector]) |handler| { handler(); // device: run the registered handler - apic.eoi(); // ...acknowledge the LAPIC } // else: spurious/unhandled — deliberately no EOI ``` @@ -100,10 +99,23 @@ Two things make device interrupts *return* where exceptions don't: flows back to `isr_common`, which restores every register it saved and executes `iretq` — resuming the interrupted instruction exactly. (This is why the stub saves *all* the general registers.) -2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss +2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss this and the LAPIC thinks we're still busy and never delivers the next interrupt. It's the single most common "my timer fired once and stopped" bug. +**Each handler issues its own EOI**, rather than the dispatcher doing it around the +call. That looks like a needless devolution while the timer is the only device, and +`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early, +because the tick hook is the scheduler, which may switch tasks and not return +promptly — the LAPIC mustn't wait on it). + +It stops looking needless with the second device. A *routed* interrupt — one arriving +through the I/O APIC from a real device line — must be **masked before it is +acknowledged**, because a level-triggered line is still asserted at EOI time and would +redeliver instantly, forever. Only the handler knows which discipline its source +needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the +device is quieted by a driver in ring 3, long after the ISR has returned. + A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so a handler must not use them; ours don't.) @@ -131,14 +143,23 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the count would stay put and the test would fail. That it advances — while the CPU was spinning in unrelated code — is the whole mechanism working end to end. +## Since (done elsewhere) + +- **Preemption**: the timer handler is where the scheduler decides to switch — the + reason a *returning* interrupt matters. See [scheduling.md](scheduling.md). +- **`sleep()` / timeouts** built on the calibrated clock. +- **The I/O APIC, routed**: external device lines now reach a vector, and the + interrupt is delivered onward to a *user-space* driver as an IPC message. See + [drivers.md](drivers.md). +- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for + user drivers — see [paging.md](paging.md). + ## What's next (not done here) -- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read - scancodes from the PS/2 controller — the first *input* device. -- **`sleep()` / timeouts** built on the calibrated clock (the monotonic - `uptimeMs()` is in place). -- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like - the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO - marked uncacheable (via the page's cache bits or an MTRR). -- **Preemption**: once there are tasks, the timer handler is where the scheduler - decides to switch — the reason a *returning* interrupt matters. +- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and ring 3 + has no port I/O yet, so the first *input* device is blocked on either an I/O + permission bitmap or `io_in`/`io_out` syscalls ([drivers.md](drivers.md)). +- **MSI/MSI-X**: per-device vectors, edge-triggered and unshared, which retire the + I/O APIC's mask/ack cycle and its 24-GSI ceiling. +- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the + identity map. QEMU tolerates it; real hardware wants it uncacheable. diff --git a/docs/driver-model.md b/docs/driver-model.md new file mode 100644 index 0000000..682bd8e --- /dev/null +++ b/docs/driver-model.md @@ -0,0 +1,305 @@ +# The driver model: buses, classes, and host controllers + +[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its +registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is +not enough for a disk, a keyboard, or a network card, because those hang off a +*controller*, on a *bus*, speaking a *protocol*, and no single process should have to +know all three. + +Real driver stacks factor into three shapes. This document is about what each one is, +what the kernel must give it, how they share code — and precisely which primitive each +is still blocked on. + +## Three shapes + +| Shape | Owns | Reaches hardware by | Talks to | +|---|---|---|---| +| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language | +| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC | +| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC | + +The last row is the surprising one and the whole point. A USB keyboard driver touches +no registers, takes no interrupts, and maps no memory. It sends HID protocol messages +to whatever published the device, and it works identically whether the controller +below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that +outlive the hardware they were written for. + +In practice **HCD and bus driver are usually the same process**. An xHCI driver is a +host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA +rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting +them is a fiction; what matters is that both *roles* have kernel support, because a +plain bus driver with no controller — a USB hub — is also a real thing. + +## The device table is the spine + +danos already has the right central structure. `src/kernel/device-service.zig` holds a table of +`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery +seeds it ([discovery.md](discovery.md)); `device_register` grows it. + +Three invariants make it a capability system rather than a directory: + +1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream — + `mmio_map`, `irq_bind`, `device_register` — checks `device_service.ownerOf(id) == me`. +2. **A descriptor is a licence to map physical memory.** Whoever claims a device may + map its `.memory` resources and bind its `.irq` resources. This is why + `device_register` cannot be a free-for-all. +3. **Therefore: containment.** Every resource of a registered child must lie inside a + resource of the same kind on its parent (`device_service.contains`). A bus driver can only + ever *subdivide* what it already holds. Without this, `device_register` would be a + syscall named "map any physical page you like." + +Containment is transitive by construction: a grandchild is contained in its child, +which is contained in the bus. Nothing can be laundered through a chain. + +Note that firmware topology does **not** obey containment, and isn't asked to — a PCI +function's BAR is not inside its host bridge's `bus_range`, because a bus-number range +is not an address window. Discovery is trusted; user space is not. + +### What a bus driver looks like + +`sbin/busd.zig` is the smallest honest one. Its "bus" is the HPET's register block and +its "devices" are the block's comparators: + +```zig +_ = dev.claim(bus.id); // 1. own the bus +const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware +const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children + +for (0..n) |i| { // 3. publish each child + var child = std.mem.zeroes(dev.DeviceDesc); + child.class = @intFromEnum(dev.DeviceClass.timer); + child.resource_count = 1; + child.resources[0] = .{ .kind = memory, + .start = bus_mmio.start + 0x100 + 0x20 * i, + .len = 0x20 }; + _ = dev.register(bus.id, &child).?; // kernel checks containment +} +``` + +Each child is left **unclaimed**, which is the handoff: a comparator driver can now +`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child +whose window escapes the bus is refused — `busd` asserts that, and the `bus` test +asserts the kernel's table upholds it. + +A USB device has *no* resources at all: `resource_count = 0`, because it's addressed +through its controller, not by MMIO. That case is allowed and is the common one. + +## Families: sharing code between drivers + +A "family" is two modules, not one: + +- **A logic module** — the parts of the bus that every driver on it re-derives. Config + space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub + protocol for USB. +- **A protocol module** — the IPC message types that let a class driver talk to + *whatever* published its device. This is the part that makes class drivers portable. + +danos already has one of each: `lib/device.zig` is a logic module, +[`lib/vfs-protocol.zig`](lib/vfs-protocol.zig) is a protocol module shared by `sbin/vfs.zig` +and its clients. The pattern generalises directly: + +``` +lib/ + rt.zig module "rt" — syscalls, heap, ipc, dev, stdio + mmio.zig module "mmio" — volatile register access + barriers [M14] + bus/ + pci.zig module "pci" — ECAM, BAR decode, capability walk + usb.zig module "usb" — descriptors, control transfers, hubs + proto/ + vfs.zig module "proto.vfs" (today: lib/vfs-protocol.zig) + block.zig module "proto.block" + hid.zig module "proto.hid" + +sbin/ + xhcid.zig HCD + bus driver imports rt, pci, usb, mmio + usbhid.zig class driver imports rt, usb, proto.hid + blockd.zig class driver imports rt, proto.block +``` + +The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one +module (`rt_mod`) and injects it. It should take a slice of modules. That's a +five-line change, and it's the *entire* mechanism — Zig modules already give you +everything else. + +The discipline that makes this work: **a class driver must not import a bus's logic +module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`. +If a class driver needs `mmio`, it has become an HCD and should be one. + +## What exists today + +- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device + grants, `device_grant` teardown. +- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before + EOI; `irq_ack` is the unmask. +- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment. + +So: **bus drivers work now.** HCDs and class drivers do not. Here is exactly why, and +exactly what would fix it. + +--- + +# Proposed ABI + +## M13 — capability passing, for class drivers + +**The blocker.** A class driver has to reach *its* device. Today the only way to find +an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`, +where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot +mint one endpoint per USB device that way, and there is no way for a bus driver to +*hand* a class driver an endpoint. M7 deferred this deliberately. + +**The fix.** Let a message carry one handle. Sender names a handle in its own table; +the kernel installs the endpoint into the receiver's table (bumping `refcount`) and +tells the receiver the index it landed at. + +``` +ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len +ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap) + -> recv_len (rax), badge (rdx), received_cap (r8) +``` + +`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred +endpoint was installed at in the receiver's table, or `no_cap`. + +- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`, + leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`; + this needs a third (`setSyscallResult3`). +- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is + not delivered** — a half-delivered capability is worse than a failed send. +- `closeHandles` already drops references on exit, so the lifetime story is unchanged. + +That single primitive gives you the standard `open` pattern: + +```zig +// class driver // bus driver +const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...); +const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint, + .{ .op = .open, .id = dev_id }); // reply with it as send_cap +// now dev_ep is a private channel to that one device +``` + +## M14 — DMA memory and the memory-ordering contract, for HCDs + +**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the +device can read, which means memory that is (a) physically contiguous, (b) at a +physical address the driver knows, (c) of the right cacheability, and (d) pinned. +[`sysMmap`](src/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()` +once per page, maps writeback-cached, and never reveals a physical address. + +**The fix.** + +``` +dma_alloc(len, flags) -> vaddr (rax), paddr (rdx) +dma_free(vaddr, len) -> 0 + +flags: dma_coherent (1) uncacheable; the default and the only one that's portable + dma_wc (2) write-combining — needs PAT programmed; for framebuffers + dma_below_4g (4) for devices with 32-bit DMA addressing +``` + +Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the +mapping, and the physical address is stable. It needs one thing the kernel lacks — +`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time +with no adjacency guarantee. + +**The memory-ordering contract.** danos has, at the time of writing, **zero memory +barriers anywhere in the tree.** That is currently correct-by-accident and won't +survive the first DMA driver, or the first ARM boot. + +`volatile` is not a barrier. In Zig it means: don't elide this access, and don't +reorder it against *other volatile* accesses. It says nothing about your *ordinary* +stores — the descriptor you just filled in normal WB memory — which LLVM may freely +sink past a volatile MMIO write. The canonical bug: + +```zig +ring[i] = descriptor; // ordinary store to WB RAM +doorbell.* = i; // volatile store to UC MMIO +// nothing stops the compiler reordering these; the device reads a stale descriptor +``` + +So the rules, which belong in `lib/mmio.zig` and behind `arch`: + +| Situation | Required | +|---|---| +| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) | +| Fill DMA descriptor, then ring doorbell | `wmb()` between them | +| Woken by IRQ, then read what the device wrote | `rmb()` before the read | +| MMIO write that must complete before the next read | `mb()` | + +And the per-arch lowering — the reason this must be an `arch` primitive and not a +sprinkling of `asm volatile`: + +| | x86_64 | aarch64 | +|---|---|---| +| `mb()` | `mfence` | `dsb sy` | +| `rmb()` | `lfence` | `dsb ld` | +| `wmb()` | `sfence` | `dsb st` | +| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance | + +x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away +with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win +condition. Build the abstraction while there is one caller to fix. + +(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full +barrier, or per-arch inline asm — which is what `lib/mmio.zig` should hide.) + +## M15 — interrupts for PCI devices + +**The blocker, and it's a hard one.** No PCI device can take an interrupt today. +[`addBars`](src/device/acpi.zig) records `.memory` and `.io_port` BARs and never an +`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpetd` only works because the +HPET advertises its own routing options in its own registers — a privilege no ordinary +device has. + +**The fix, in two halves.** + +*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it +as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**, +and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the +line must be polled on each interrupt — the reason everyone left INTx behind. + +*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no +mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver +the (address, data) pair to program into its own MSI capability: + +``` +msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 } +``` + +The driver writes those into config space itself — which means it needs config space, +which means **discovery should give each `pci_device` a `.memory` resource for its +4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole +capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall. + +Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpetd` can never +exercise this path. The first MSI driver will be the first PCI driver. + +## M16 — the IOMMU, and the honest caveat + +Everything above is capability-gated at the *CPU*. None of it is gated at the *device*. +A driver that can program a bus-mastering engine can make that device write to any +physical address, because page tables sit between the CPU and RAM, not between a device +and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim` +on any DMA-capable device is equivalent to granting ring 0.** + +This does not make the model useless — it's the same position Linux is in with the +IOMMU off, and every other guarantee (crash isolation, restart, no shared address +space) still holds. But "user-space drivers are memory-safe" is not true yet, and the +gap should be named rather than implied. + +## Ordering + +`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It +unlocks class drivers, which are the shape with no hardware requirements at all — you +could write a real one against `busd`'s comparators tomorrow. + +`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing +on its own regardless: it's small, obviously correct, and stops every future driver +from hand-rolling `*volatile` and getting ARM wrong. + +## See also + +- [drivers.md](drivers.md) — how to write one, concretely. +- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from. +- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on. +- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble. diff --git a/docs/drivers.md b/docs/drivers.md new file mode 100644 index 0000000..bb0c1f2 --- /dev/null +++ b/docs/drivers.md @@ -0,0 +1,319 @@ +# Writing a driver + +In a monolithic kernel a driver is a function call away from everything: it runs in +ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In +danos a driver is **an ordinary ring-3 process**. It has its own address space, it +can crash without taking the kernel with it, and — the point of this document — it +can be restarted ([resilience](resilience.md)). + +That leaves three questions the kernel has to answer, because a process can't answer +them for itself: + +1. **What hardware exists?** → `device_enumerate`, over the device table discovery built + ([discovery](discovery.md), [acpi](acpi.md)). +2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the + device's physical MMIO window into your address space, and from then on it's plain + memory. No syscall per register access. +3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered + to you as an IPC notification. You block; the hardware wakes you. + +A driver is, in one sentence, *a process that sleeps until its device has something to +say.* + +## The capability: claim before touch + +The five driver syscalls (`src/root.zig`, dispatched in `src/kernel/process.zig`): + +| # | Call | Meaning | +|---|------|---------| +| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table | +| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership | +| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window | +| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification | +| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device | +| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed | + +Notice that **nothing takes a physical address or an interrupt number.** Every call +names a device by id and a resource by index. That indirection is the entire security +model. If `mmio_map` took a physical address, any process could map the kernel's +memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and +silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and +the same check at the top of `sysMmioMap`): + +- `device_service.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive +- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq` + for `irq_bind` + +The claim is the capability. Everything else follows from it. + +## Registers: `mmio_map` + +`mmio_map` walks the caller's page tables and installs the device's physical frames +with `present | user | writable | nx | pcd | pwt` +(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing: + +- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read + would return a stale value and a write might never leave the CPU. +- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO + rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is + destroyed. Without this, killing a driver would hand the HPET's registers back to + the frame allocator as if they were free RAM. The `iopass` test guards it. + +Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages +never widen an existing mapping. + +Then you just… use it: + +```zig +const base = dev.mmioMap(dev_id, mmio_res) orelse return; +const counter: *volatile u64 = @ptrFromInt(base + 0xF0); +const now = counter.*; // a load, straight to the hardware. no kernel involved. +``` + +## Interrupts: the cycle, and why it has that shape + +An interrupt handler in a microkernel has a problem. The code that knows how to quiet +the device is in ring 3, in another address space, and it will not run for +microseconds or milliseconds — after a context switch, when the scheduler gets to it. +But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until +the device is quieted. EOI a still-asserted line and the I/O APIC redelivers +immediately. Forever. The driver never gets to run at all. + +The way out is to mask the line before acknowledging it: + +``` +kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU + irqEoi() // now safe to tell the LAPIC we're done + notifyFromIsr() // wake the driver — it runs much later + +driver replyWait() -> badge with the notify bit set + // NOW the line deasserts + irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire + +``` + +`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the +interrupt fires exactly once, ever; call it before the device is quiet and you get an +interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two +syscalls and not one. + +This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI +itself. It used to, before running the handler — correct for the LAPIC timer, and +impossible for a routed device line. Each handler now owns its EOI, because only the +handler knows which discipline its source needs. + +### The driver side is an event loop, not a callback + +`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by +the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one +single-threaded loop over both of its event sources: + +```zig +while (true) { + const r = ipc.replyWait(endpoint, reply, &recv); + if (r.isNotification()) { // r.source() is the GSI + service_device(); // clear the status register + _ = dev.irqAck(id, irq_res); // re-arm + } else { + handle_client_request(recv[0..r.len]); + } +} +``` + +No reentrancy, no "what am I allowed to call from an interrupt handler", no shared +state between ISR and task context. The interrupt is just a message. + +Two properties worth knowing: + +- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in + an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody + waiting, but the badge is already on the endpoint's notify ring. The next + `replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO). +- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on + overflow. That's correct: an IRQ notification is a *level* ("the device wants + attention"), not a tally. Re-read the device's status register; never assume one + notification means exactly one event. Because the ISR masks the line until you + `irq_ack`, at most one badge per GSI can be outstanding — so the ring can only + overflow if you bind more than eight GSIs to a single endpoint. Don't. + +## A whole driver + +`sbin/hpetd.zig` is ~150 lines and does all of it. The shape: + +```zig +const hpet = findHpet(buf) orelse return; // device_enumerate, look for + // class=timer with memory + irq +_ = dev.claim(hpet.dev_id); // the capability +const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?; +const endpoint = ipc.createEndpoint().?; + +// program the hardware over the mapping we were just handed +reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config +reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator +reg(base, 0x010).* |= 1; // ENABLE + +_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint); + +while (...) { + const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling. + if (r.badge & notify_bit == 0) continue; + reg(base, 0x020).* = 1; // clear status -> deassert + reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm + _ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask +} +``` + +The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a +clocksource — the only way to use it is to read it, so it proved `mmio_map` without +needing interrupts at all. Its *comparators* are a clockevent, and can be configured +**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS` +that the driver must write-1-to-clear. That's a genuine deassert step, so the full +mask/ack cycle above is exercised for real rather than being decoration on an +edge-triggered line that would have been fine without it. + +One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**. +Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in +the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the +mask, and records one concrete GSI as an `irq` resource. The driver then programs +`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one +it recorded. Hardware that describes itself at runtime still has to fit through a +static capability. + +## Publishing children: `device_register` + +A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block +of comparators — needs a driver that enumerates it and tells the kernel what it found. +That's `device_register`, and it makes the device table a tree rather than a list +(`DeviceDesc.parent`). + +```zig +var child = std.mem.zeroes(dev.DeviceDesc); +child.class = @intFromEnum(dev.DeviceClass.timer); +child.resource_count = 1; +child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 }; +const child_id = dev.register(bus_id, &child).?; +``` + +The child is left **unclaimed**, which is the whole point: another process claims it and +`mmio_map`s it, and sees only that 0x20-byte window. + +The rule the kernel enforces is **containment**: every resource of a child must lie +inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match +exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so +without containment `device_register` would be a syscall for mapping any page you like. A +bus driver may only ever subdivide what it already owns. + +A device with **no resources** is legal and common. A USB device is reached through its +controller, not by MMIO, so it gets `resource_count = 0`. + +See [`sbin/busd.zig`](../sbin/busd.zig) for a complete one, and +[driver-model.md](driver-model.md) for how bus drivers, class drivers and host +controller drivers fit together. + +## What the kernel does not do for you + +- **It does not quiet your device.** That's the whole reason `irq_ack` exists. +- **It does not know your registers.** `mmio_map` hands you a base address; every + offset in this document came from the HPET spec, not from danos. +- **It does not serialise your driver.** Two clients calling one driver endpoint are + serialised by `replyWait`, but nothing stops your driver from being preempted. + +## Limits, today + +Worth knowing before you write the second driver: + +- **Ring 3 has no port I/O.** The TSS I/O permission bitmap is absent + (`tss.zig`: `iomap_base = @sizeOf(Tss)`), and IOPL is never raised, so `in`/`out` + from a driver is a #GP. That rules out a user-space 16550 UART (`0x3F8`), PS/2 + (`0x60`/`0x64`), and legacy PCI config (`0xCF8`/`0xCFC`). Everything must be MMIO. + `io_port` resources are recorded by discovery and then ignored. +- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means + granting one grants the other. A `device_register`ed child's *resource* can be narrower + than a page, but its *mapping* can't. +- **No DMA memory.** `mmap` gives you writeback-cached, non-contiguous pages and never + tells you their physical address, so you cannot build a descriptor ring. Any driver + for a bus-mastering device is blocked on this. +- **No memory barriers.** There are none in the tree, and `volatile` is not one — it + won't stop the compiler sinking an ordinary store (your DMA descriptor) past a + volatile MMIO store (your doorbell). On x86 you mostly get away with it; on ARM you + will not. See [driver-model.md](driver-model.md#m14). +- **DMA is not contained.** A driver that can program a bus-mastering device can make + that device write to *any* physical address — page tables don't sit between a device + and RAM; an IOMMU does. Until VT-d/DMAR is programmed, `device_claim` on a DMA-capable + device is effectively equivalent to granting ring 0. This is the largest gap between + the design's promise and what it delivers. +- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two + drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real + answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`). +- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override + says active-low needs that threaded through from discovery. +- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and + by a single I/O APIC. +- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops + on overflow. With one GSI per endpoint that's unreachable — the line is masked from + the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices + to one endpoint, though, and a dropped badge leaves that line masked with nobody + left to ack it. +- **A faulting driver still kills the machine.** There is no per-process kill path: a + ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary + `exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is + [not built yet](resilience.md). +- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the + line on exit, but the claim is never released — restart is + [not built](resilience.md). +- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a + level-triggered redirection entry with remote-IRR set doesn't clear it on some + chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the + tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and + back. See the note at the top of `src/kernel/irq.zig`. + +## Verifying it + +The `hpet` test spawns `hpetd` from the initrd and watches the serial log. The driver +prints `hpetd: ok` only after being woken five times, and its loop's only exit is +through `replyWait` returning a notification — it cannot reach that line by polling. + +The last check doesn't trust the driver's self-report at all: the kernel reads the I/O +APIC redirection entry back and asserts the line really is routed to a device vector, +really is level-triggered, and really was left unmasked by the driver's final +`irq_ack`. + +``` +$ python3 test/qemu_test.py hpet irqfree iopass + hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS') + irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS') + iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS') +``` + +Two companions cover what `hpetd` can't, because it never exits: + +- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases + one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's + is not. That second half is why bindings are keyed on the owning *task* and not on + the endpoint pointer — endpoints are shared, so releasing "everything pointing at + this endpoint" would silently mask a live driver's device. +- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address + space never returns MMIO frames to the RAM pool. + +## What's next (not done here) + +The big ones — capability passing (class drivers), DMA + barriers and MSI (host +controller drivers), and the IOMMU — have proposed signatures in +[driver-model.md](driver-model.md). Smaller items: + +- **Port I/O grants**, so a PS/2 or 16550 driver is possible: either a per-device TSS + I/O permission bitmap swapped on context switch, or `io_in`/`io_out` syscalls gated + by the same claim. The legacy devices that need it are all low-rate, so the syscall + is likely fast enough. +- **Releasing a claim.** There is no `dev_release`, and `device_service` never drops a claim on + exit — only IRQ bindings are released. A dead driver's device stays owned forever, + which blocks restart. +- **Unregistering children.** `device_register` only appends. A USB device that is + unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry + table. +- **Restart.** A driver that dies should release its claim, have its device quiesced, + and be respawned by a supervisor. Some pieces (`releaseIrqs`, `device_grant` + teardown, the claim table) exist; the policy doesn't. +- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken + driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so + a woken driver waits for the next scheduling point. diff --git a/docs/ipc.md b/docs/ipc.md index 3464ce3..e58a3e2 100644 --- a/docs/ipc.md +++ b/docs/ipc.md @@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what was a function call across a monolithic kernel is IPC, so it's a first-class concern, not an afterthought. -This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size -ring buffer of messages with a producer/consumer rendezvous, built on the -scheduler's [wait queues](scheduling.md). +There are two layers, built a milestone apart: + +- **`src/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*, + described below. The primitive, and where the blocking discipline was worked out. +- **`src/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across + address spaces. What user-space servers and drivers actually talk over. It's the + second half of this document. ## The channel +The first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size +ring buffer of messages with a producer/consumer rendezvous, built on the +scheduler's [wait queues](scheduling.md). + `Channel(T, capacity)` is generic over the message type and buffer size. It holds a ring buffer, a count, and two wait queues: @@ -42,16 +50,51 @@ full and empty over and over, so both the blocking-send and blocking-recv paths exercised heavily. The messages arrive intact and in order (their sum is the expected `5050`), and neither task busy-waits — they block and wake each other. +## Endpoints: call/reply across address spaces + +A channel connects two kernel threads sharing one address space. Real servers are +*processes*, so the payload has to cross an address-space boundary. That's +`src/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an +`Endpoint`, with the message copied directly from the sender's pages to the receiver's +(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no +bounce buffer). + +Two syscalls carry it: + +- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies. +- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if + any), then block for the next request. One syscall, because a server's steady state + is *always* "finish the last one, wait for the next". + +An endpoint is reached by **handle** — a small integer index into the process's handle +table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The +bootstrap problem (how do you get the first handle?) is solved by a tiny name registry: +a server calls `ipc_register(service_id, h)` under a well-known small integer, and a +client calls `ipc_lookup(service_id)`. + +The server never learns the client's identity beyond a **badge**, delivered alongside +the message: the caller's task id. + +### Interrupts are messages too + +`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no +reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit +set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a +client wants something" from "the hardware wants something". Notifications sit in a +small coalescing ring on the endpoint, so an interrupt taken while the driver was busy +elsewhere is not lost. + +This is what makes a user-space driver possible at all, and it's the subject of +[drivers.md](drivers.md). + ## What's next (not done here) -- **Across address spaces.** Today both endpoints are kernel threads sharing the - kernel's memory, so the message is copied within one address space. When user - mode arrives, the same channel carries messages between *isolated* processes, - copying the payload across the boundary — which is where IPC earns its place as - the microkernel's backbone. -- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply) - on top of channels, the shape most driver/service calls take. -- **Interrupts as messages.** A hardware interrupt delivered to the driver task - that owns the device, as an IPC message. - **Priority inheritance** through IPC, so a high-priority client blocked on a low-priority server doesn't suffer unbounded priority inversion. +- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so + every capability is either well-known (the registry) or inherited — there's no way + to delegate one. +- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong + shape (logging, notifications between servers). +- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel + lock; a bulk transfer wants shared pages, not a copy. diff --git a/lib/dev.zig b/lib/dev.zig deleted file mode 100644 index 9b7a75f..0000000 --- a/lib/dev.zig +++ /dev/null @@ -1,32 +0,0 @@ -//! User-space device access: enumerate the kernel's device table, claim a -//! device, and map its MMIO. A driver uses these to find and take ownership of -//! its hardware; the claim is the capability the kernel checks before mapping. - -const danos = @import("danos"); -const sc = @import("syscall.zig"); - -pub const DeviceDesc = danos.DeviceDesc; -pub const ResDesc = danos.ResDesc; -pub const DeviceClass = danos.DeviceClass; -pub const ResourceKind = danos.ResourceKind; - -inline fn failed(r: usize) bool { - return r > ~@as(usize, 0) - 4095; -} - -/// Copy up to `buf.len` device descriptors into `buf`; returns the total count. -pub fn enumerate(buf: []DeviceDesc) usize { - return sc.syscall2(.dev_enumerate, @intFromPtr(buf.ptr), buf.len); -} - -/// Take exclusive ownership of device `id`. Returns false if taken or invalid. -pub fn claim(id: u64) bool { - return !failed(sc.syscall1(.dev_claim, id)); -} - -/// Map resource `res_idx` (which must be an MMIO window) of claimed device -/// `dev_id` into this address space; returns the register base virtual address. -pub fn mmioMap(dev_id: u64, res_idx: u64) ?usize { - const r = sc.syscall2(.mmio_map, dev_id, res_idx); - return if (failed(r)) null else r; -} diff --git a/lib/device.zig b/lib/device.zig new file mode 100644 index 0000000..9a914f8 --- /dev/null +++ b/lib/device.zig @@ -0,0 +1,67 @@ +//! User-space device access: enumerate the kernel's device table, claim a device, +//! map its MMIO, and bind its interrupt. A driver uses these to find and take +//! ownership of its hardware; the claim is the capability the kernel checks before +//! mapping registers or routing an IRQ. + +const danos = @import("danos"); +const sc = @import("system-call.zig"); + +pub const DeviceDescriptor = danos.DeviceDescriptor; +pub const ResourceDescriptor = danos.ResourceDescriptor; +pub const DeviceClass = danos.DeviceClass; +pub const ResourceKind = danos.ResourceKind; + +inline fn failed(r: usize) bool { + return r > ~@as(usize, 0) - 4095; +} + +/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count. +pub fn enumerate(buffer: []DeviceDescriptor) usize { + return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len); +} + +/// Take exclusive ownership of device `id`. Returns false if taken or invalid. +pub fn claim(id: u64) bool { + return !failed(sc.systemCall1(.device_claim, id)); +} + +/// Map resource `resource_index` (which must be an MMIO window) of claimed device +/// `device_id` into this address space; returns the register base virtual address. +pub fn mmioMap(device_id: u64, resource_index: u64) ?usize { + const r = sc.systemCall2(.mmio_map, device_id, resource_index); + return if (failed(r)) null else r; +} + +/// `DeviceDescriptor.parent` for a device with no parent. +pub const no_parent = danos.no_parent; + +/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed. +/// Returns the new device id. The child is left unclaimed, so whichever driver owns +/// that class of device can `claim` it — that is how a bus hands off a device. +/// +/// Every resource in `descriptor` must be **contained** in a parent resource of the same +/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses +/// anything else, because a device descriptor is a licence to map physical memory and +/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent` +/// are ignored. A device with no resources at all is fine — a USB device is reached +/// through its controller, not by MMIO. +pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 { + const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor)); + return if (failed(r)) null else r; +} + +/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to +/// `endpoint`. From then on the interrupt arrives as an asynchronous notification: +/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the +/// low bits carrying the GSI. The kernel masks the line before waking you. +pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool { + return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint)); +} + +/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever +/// status register holds its line asserted) — the kernel left the line masked +/// precisely because it could not do that for you. Skip it and the interrupt never +/// fires again; call it before the device is quiet and a level-triggered line storms. +pub fn irqAck(device_id: u64, resource_index: u64) bool { + return !failed(sc.systemCall2(.irq_ack, device_id, resource_index)); +} diff --git a/lib/heap.zig b/lib/heap.zig index 133efb5..aa19715 100644 --- a/lib/heap.zig +++ b/lib/heap.zig @@ -5,16 +5,16 @@ //! The algorithm is a straight port of the kernel's first-fit free list //! (src/kernel/heap.zig): an address-ordered singly linked list of free blocks, //! split on allocation and coalesced with neighbours on free. The only thing -//! that changes on this side of the syscall boundary is where memory comes from +//! that changes on this side of the system_call boundary is where memory comes from //! — `grow` asks the kernel for pages via `mmap` instead of mapping frames //! itself, and the kernel picks the base address. //! -//! Single-threaded and 16-byte max alignment, exactly like the kernel heap; a +//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a //! lock and larger alignments come when user programs gain threads. const std = @import("std"); const danos = @import("danos"); -const sys = @import("sys.zig"); +const system = @import("system.zig"); const page_size = danos.page_size; @@ -26,8 +26,8 @@ const Block = extern struct { }; const header_size = @sizeOf(Block); // 16 -const min_block = header_size + 16; // smallest block worth splitting off -/// Grow granularity: one `mmap` per 64 KiB amortises the syscall. +const minimum_block = header_size + 16; // smallest block worth splitting off +/// Grow granularity: one `mmap` per 64 KiB amortises the system_call. const chunk = 64 * 1024; var free_list: ?*Block = null; @@ -44,10 +44,10 @@ fn payloadOf(block: *Block) [*]u8 { /// `mmap` is an independent grant, cross-grant coalescing happens only when the /// kernel returns adjacent bases (its arena is a bump allocator, so consecutive /// grants usually are adjacent). Returns false if the kernel is out of memory. -fn grow(min_bytes: usize) bool { - const bytes = alignUp(@max(min_bytes, chunk), page_size); - const ret = sys.mmap(bytes, sys.PROT_READ | sys.PROT_WRITE); - if (sys.mmapFailed(ret)) return false; +fn grow(minimum_bytes: usize) bool { + const bytes = alignUp(@max(minimum_bytes, chunk), page_size); + const ret = system.mmap(bytes, system.PROT_READ | system.PROT_WRITE); + if (system.mmapFailed(ret)) return false; const block: *Block = @ptrFromInt(ret); block.size = bytes; @@ -58,25 +58,25 @@ fn grow(min_bytes: usize) bool { /// Insert a block into the address-ordered free list, coalescing with the /// physically adjacent free blocks on either side. fn insertFree(block: *Block) void { - var prev: ?*Block = null; - var cur = free_list; - while (cur) |c| : (cur = c.next) { + var previous: ?*Block = null; + var current = free_list; + while (current) |c| : (current = c.next) { if (@intFromPtr(c) > @intFromPtr(block)) break; - prev = c; + previous = c; } - block.next = cur; - if (prev) |p| p.next = block else free_list = block; + block.next = current; + if (previous) |p| p.next = block else free_list = block; - // Merge forward into `cur` if they're contiguous. - if (cur) |c| { + // Merge forward into `current` if they're contiguous. + if (current) |c| { if (@intFromPtr(block) + block.size == @intFromPtr(c)) { block.size += c.size; block.next = c.next; } } - // Merge `prev` forward into `block` if they're contiguous. - if (prev) |p| { + // Merge `previous` forward into `block` if they're contiguous. + if (previous) |p| { if (@intFromPtr(p) + p.size == @intFromPtr(block)) { p.size += block.size; p.next = block.next; @@ -90,24 +90,24 @@ fn rawAlloc(len: usize) ?[*]u8 { var attempts: u32 = 0; while (attempts < 2) : (attempts += 1) { - var prev: ?*Block = null; - var cur = free_list; - while (cur) |block| : ({ - prev = block; - cur = block.next; + var previous: ?*Block = null; + var current = free_list; + while (current) |block| : ({ + previous = block; + current = block.next; }) { if (block.size < need) continue; - if (block.size >= need + min_block) { + if (block.size >= need + minimum_block) { // Split: carve `need` off the front, leave the rest free. const rest: *Block = @ptrFromInt(@intFromPtr(block) + need); rest.size = block.size - need; rest.next = block.next; - if (prev) |p| p.next = rest else free_list = rest; + if (previous) |p| p.next = rest else free_list = rest; block.size = need; } else { // Take the whole block. - if (prev) |p| p.next = block.next else free_list = block.next; + if (previous) |p| p.next = block.next else free_list = block.next; } return payloadOf(block); } diff --git a/lib/ipc.zig b/lib/ipc.zig index 0f1545c..d28e97b 100644 --- a/lib/ipc.zig +++ b/lib/ipc.zig @@ -4,7 +4,7 @@ //! added with the first server binary. const danos = @import("danos"); -const sc = @import("syscall.zig"); +const sc = @import("system-call.zig"); /// A small-int handle into the calling process's handle table. pub const Handle = usize; @@ -18,61 +18,77 @@ pub const Message = extern struct { c: u64 = 0, }; -/// Whether a syscall return value is a wrapped -errno (lands in the top page). +/// Whether a system_call return value is a wrapped -errno (lands in the top page). inline fn failed(r: usize) bool { return r > ~@as(usize, 0) - 4095; } /// Create a new endpoint owned by this process; returns its handle. pub fn createEndpoint() ?Handle { - const r = sc.syscall0(.create_endpoint); + const r = sc.systemCall0(.create_endpoint); return if (failed(r)) null else r; } /// Publish endpoint `h` under a well-known service id so other processes find it. pub fn register(id: danos.ServiceId, h: Handle) bool { - return !failed(sc.syscall2(.ipc_register, @intFromEnum(id), h)); + return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h)); } /// Find the endpoint published under `id`, installing a handle to it in this /// process. pub fn lookup(id: danos.ServiceId) ?Handle { - const r = sc.syscall1(.ipc_lookup, @intFromEnum(id)); + const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id)); return if (failed(r)) null else r; } pub const CallError = error{Failed}; -/// Send `msg` to endpoint `h` and block until the server replies into `reply`. +/// Send `message` to endpoint `h` and block until the server replies into `reply`. /// Returns the reply length. -pub fn call(h: Handle, msg: []const u8, reply: []u8) CallError!usize { - const r = sc.syscall5(.ipc_call, h, @intFromPtr(msg.ptr), msg.len, @intFromPtr(reply.ptr), reply.len); +pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize { + const r = sc.systemCall5(.ipc_call, h, @intFromPtr(message.ptr), message.len, @intFromPtr(reply.ptr), reply.len); return if (failed(r)) error.Failed else r; } +/// Set in `Received.badge` when what arrived is an asynchronous notification — a +/// bound device interrupt — rather than a client's message. The low bits carry the +/// GSI. See `isNotification`. +pub const notify_badge_bit: u64 = danos.notify_badge_bit; + /// The result of a `replyWait`: the request length and the sender's badge (a /// task id, or an IRQ notification if the high bit is set). pub const Received = struct { len: usize, badge: u64, + + /// True if this wake-up was a device interrupt, not a client request. A driver's + /// event loop branches on this; there is no reply owed on the notification path. + pub fn isNotification(self: Received) bool { + return self.badge & notify_badge_bit != 0; + } + + /// The interrupt source (a GSI), meaningful only when `isNotification`. + pub fn source(self: Received) u64 { + return self.badge & ~notify_badge_bit; + } }; /// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if -/// any), then block until the next request arrives in `recv`. Returns its length -/// and the sender badge. This syscall returns two values — the length in rax and +/// any), then block until the next request arrives in `receive`. Returns its length +/// and the sender badge. This system_call returns two values — the length in rax and /// the badge in rdx — so it needs a hand-written stub: rdx is a read-write /// operand (input = reply length, arg #3; output = badge). -pub fn replyWait(h: Handle, reply: []const u8, recv: []u8) Received { +pub fn replyWait(h: Handle, reply: []const u8, receive: []u8) Received { var rax: usize = undefined; var rdx: usize = reply.len; // in: reply_len (arg #3 -> rdx); out: badge asm volatile ("syscall" : [rax] "={rax}" (rax), [rdx] "+{rdx}" (rdx), - : [n] "{rax}" (@intFromEnum(danos.Syscall.ipc_reply_wait)), + : [n] "{rax}" (@intFromEnum(danos.SystemCall.ipc_reply_wait)), [a0] "{rdi}" (h), [a1] "{rsi}" (@intFromPtr(reply.ptr)), - [a3] "{r10}" (@intFromPtr(recv.ptr)), - [a4] "{r8}" (recv.len), + [a3] "{r10}" (@intFromPtr(receive.ptr)), + [a4] "{r8}" (receive.len), : .{ .rcx = true, .r11 = true, .memory = true }); return .{ .len = rax, .badge = rdx }; } diff --git a/lib/rt.zig b/lib/runtime.zig similarity index 66% rename from lib/rt.zig rename to lib/runtime.zig index 27f14a3..aef0034 100644 --- a/lib/rt.zig +++ b/lib/runtime.zig @@ -1,29 +1,29 @@ //! danos user-space runtime library — a nascent libc. Every user binary (init, -//! and later the VFS server + device drivers) imports this as `@import("rt")`: -//! syscall wrappers, the C-convention heap, IPC helpers, and the process start +//! and later the VFS server + device drivers) imports this as `@import("runtime")`: +//! system_call wrappers, the C-convention heap, IPC helpers, and the process start //! shim. It is compiled into each binary (inheriting its `.large` code model and //! freestanding target), so all user programs share one implementation. //! //! A user binary needs three lines: -//! const rt = @import("rt"); -//! pub const panic = rt.panic; -//! comptime { _ = &rt.start._start; } // pull the entry shim in +//! const runtime = @import("runtime"); +//! pub const panic = runtime.panic; +//! comptime { _ = &runtime.start._start; } // pull the entry shim in //! and a `pub fn main() void`. -pub const sys = @import("sys.zig"); +pub const system = @import("system.zig"); pub const heap = @import("heap.zig"); pub const ipc = @import("ipc.zig"); pub const start = @import("start.zig"); /// The VFS wire protocol (shared with the VFS server). -pub const vfsproto = @import("vfs_proto.zig"); +pub const vfs_protocol = @import("vfs-protocol.zig"); /// POSIX-style file API: open/read/write/lseek/stat/close. pub const unistd = @import("unistd.zig"); /// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd. pub const stdio = @import("stdio.zig"); /// Device access for drivers: enumerate/claim/mmioMap. -pub const dev = @import("dev.zig"); +pub const device = @import("device.zig"); -/// Re-exported so a user binary can `pub const panic = rt.panic;`. +/// Re-exported so a user binary can `pub const panic = runtime.panic;`. pub const panic = start.panic; /// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code. diff --git a/lib/start.zig b/lib/start.zig index 15a4892..bd6bdfd 100644 --- a/lib/start.zig +++ b/lib/start.zig @@ -1,11 +1,11 @@ //! The user-space process entry shim. Every user binary roots `_start` here (via //! `entry = _start` in build.zig) and forces this file to be analysed with -//! `comptime { _ = &rt.start._start; }`, so the whole runtime is linked in. +//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in. const std = @import("std"); -const sys = @import("sys.zig"); +const system = @import("system.zig"); -/// The kernel enters at `_start` with rsp 16-aligned, but a SysV function expects +/// The kernel enters at `_start` with rsp 16-aligned, but a SystemV function expects /// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes /// the 8-byte return address, satisfying the ABI before any Zig frame runs; the /// `ud2` is a safety net if `rt_start` ever returns. @@ -21,12 +21,12 @@ pub export fn _start() callconv(.naked) noreturn { export fn rt_start() callconv(.c) noreturn { const root = @import("root"); // the user binary's root source file root.main(); - sys.exit(0); + system.exit(0); } /// No runtime to unwind into — report a panic as a nonzero exit code. pub const panic = std.debug.FullPanic(struct { fn panic(_: []const u8, _: ?usize) noreturn { - sys.exit(127); + system.exit(127); } }.panic); diff --git a/lib/stdio.zig b/lib/stdio.zig index 8d8155a..ddc32ac 100644 --- a/lib/stdio.zig +++ b/lib/stdio.zig @@ -8,7 +8,7 @@ const unistd = @import("unistd.zig"); const heap = @import("heap.zig"); pub const SEEK_SET = unistd.SEEK_SET; -pub const SEEK_CUR = unistd.SEEK_CUR; +pub const SEEK_CURRENT = unistd.SEEK_CURRENT; pub const SEEK_END = unistd.SEEK_END; /// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the @@ -47,10 +47,10 @@ pub fn fclose(f: *FILE) c_int { } /// Read `size*nmemb` bytes; returns the number of whole items read. -pub fn fread(buf: []u8, size: usize, nmemb: usize, f: *FILE) usize { +pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize { const total = size * nmemb; if (total == 0) return 0; - const n = unistd.read(f.fd, buf[0..@min(buf.len, total)]); + const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]); if (n <= 0) { f.eof = 1; return 0; @@ -76,7 +76,7 @@ pub fn fseek(f: *FILE, off: i64, whence: u32) c_int { } pub fn ftell(f: *FILE) i64 { - return unistd.lseek(f.fd, 0, unistd.SEEK_CUR); + return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT); } pub fn rewind(f: *FILE) void { diff --git a/lib/syscall.zig b/lib/system-call.zig similarity index 71% rename from lib/syscall.zig rename to lib/system-call.zig index 8152782..9615119 100644 --- a/lib/syscall.zig +++ b/lib/system-call.zig @@ -1,50 +1,50 @@ -//! Raw `syscall` instruction wrappers for user space — one per arity. +//! Raw `system_call` instruction wrappers for user space — one per arity. //! //! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax. -//! The `syscall` instruction itself clobbers rcx (it holds the return rip) and +//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and //! r11 (the saved rflags); the kernel entry stub preserves everything else. //! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the //! instruction, so the kernel reads the 4th argument from r10. const danos = @import("danos"); -const Syscall = danos.Syscall; +const SystemCall = danos.SystemCall; -pub inline fn syscall0(n: Syscall) usize { +pub inline fn systemCall0(n: SystemCall) usize { return asm volatile ("syscall" : [ret] "={rax}" (-> usize), : [n] "{rax}" (@intFromEnum(n)), : .{ .rcx = true, .r11 = true, .memory = true }); } -pub inline fn syscall1(n: Syscall, a0: usize) usize { +pub inline fn systemCall1(n: SystemCall, a0: usize) usize { return asm volatile ("syscall" : [ret] "={rax}" (-> usize), : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), : .{ .rcx = true, .r11 = true, .memory = true }); } -pub inline fn syscall2(n: Syscall, a0: usize, a1: usize) usize { +pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize { return asm volatile ("syscall" : [ret] "={rax}" (-> usize), : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), : .{ .rcx = true, .r11 = true, .memory = true }); } -pub inline fn syscall3(n: Syscall, a0: usize, a1: usize, a2: usize) usize { +pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize { return asm volatile ("syscall" : [ret] "={rax}" (-> usize), : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), : .{ .rcx = true, .r11 = true, .memory = true }); } -pub inline fn syscall4(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize) usize { +pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize { return asm volatile ("syscall" : [ret] "={rax}" (-> usize), : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), : .{ .rcx = true, .r11 = true, .memory = true }); } -pub inline fn syscall5(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize { +pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize { return asm volatile ("syscall" : [ret] "={rax}" (-> usize), : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), [a4] "{r8}" (a4), diff --git a/lib/sys.zig b/lib/system.zig similarity index 73% rename from lib/sys.zig rename to lib/system.zig index 06a0ead..19b9c2e 100644 --- a/lib/sys.zig +++ b/lib/system.zig @@ -1,9 +1,9 @@ -//! Typed syscall surface for user space — thin wrappers over the raw `syscall` -//! stubs, one per kernel call. Numbers come from `danos.Syscall`, the single +//! Typed system_call surface for user space — thin wrappers over the raw `system_call` +//! stubs, one per kernel call. Numbers come from `danos.SystemCall`, the single //! source of truth shared with the kernel dispatcher. const danos = @import("danos"); -const sc = @import("syscall.zig"); +const sc = @import("system-call.zig"); /// `mmap` protection flags (matching the usual C bit values). Grants are always /// readable+writable today; the kernel does not yet honour finer prot. @@ -13,23 +13,23 @@ pub const PROT_EXEC: usize = danos.prot_exec; /// Give up the rest of this quantum. pub fn yield() void { - _ = sc.syscall0(.yield); + _ = sc.systemCall0(.yield); } /// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes /// through the console/VFS later). Returns the byte count, or a wrapped -1. -pub fn write(msg: []const u8) usize { - return sc.syscall2(.debug_write, @intFromPtr(msg.ptr), msg.len); +pub fn write(message: []const u8) usize { + return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len); } /// Block the caller for `ms` milliseconds. pub fn sleep(ms: usize) void { - _ = sc.syscall1(.sleep, ms); + _ = sc.systemCall1(.sleep, ms); } /// End the process. Never returns. pub fn exit(code: usize) noreturn { - _ = sc.syscall1(.exit, code); + _ = sc.systemCall1(.exit, code); unreachable; // the kernel never returns from exit } @@ -37,12 +37,12 @@ pub fn exit(code: usize) noreturn { /// memory and return the base virtual address. On failure returns a value in the /// top page (see `mmapFailed`). The user heap grows through this call. pub fn mmap(len: usize, prot: usize) usize { - return sc.syscall2(.mmap, len, prot); + return sc.systemCall2(.mmap, len, prot); } /// Release a range previously handed out by `mmap`. pub fn munmap(base: usize, len: usize) usize { - return sc.syscall2(.munmap, base, len); + return sc.systemCall2(.munmap, base, len); } /// Whether an `mmap` return value is an error (the kernel returns a wrapped diff --git a/lib/unistd.zig b/lib/unistd.zig index 6e162c2..4268d62 100644 --- a/lib/unistd.zig +++ b/lib/unistd.zig @@ -4,13 +4,13 @@ //! kernel knows nothing of files or fds — the fd table lives here, per process. const std = @import("std"); -const proto = @import("vfs_proto.zig"); +const protocol = @import("vfs-protocol.zig"); const ipc = @import("ipc.zig"); const danos = @import("danos"); -pub const O_CREAT = proto.O_CREAT; +pub const O_CREAT = protocol.O_CREAT; pub const SEEK_SET: u32 = 0; -pub const SEEK_CUR: u32 = 1; +pub const SEEK_CURRENT: u32 = 1; pub const SEEK_END: u32 = 2; // Resolve (and cache) the VFS server endpoint, looked up by well-known id. @@ -24,9 +24,9 @@ fn vfs() ?usize { return vfs_handle; } -const max_fds = 32; +const maximum_fds = 32; const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 }; -var fds = [_]Fd{.{}} ** max_fds; +var fds = [_]Fd{.{}} ** maximum_fds; fn allocFd() ?usize { for (&fds, 0..) |*f, i| { @@ -38,30 +38,30 @@ fn allocFd() ?usize { return null; } -const Result = struct { reply: proto.Reply, payload: []u8 }; +const Result = struct { reply: protocol.Reply, payload: []u8 }; /// One request/reply round trip: [Request header][send payload] -> VFS -> -/// [Reply header][recv payload]. The recv payload is written into `out`. -fn transact(req: proto.Request, send: []const u8, out: []u8) ?Result { +/// [Reply header][receive payload]. The receive payload is written into `out`. +fn transact(req: protocol.Request, send: []const u8, out: []u8) ?Result { const h = vfs() orelse return null; - var msg: [proto.msg_max]u8 = undefined; - @memcpy(msg[0..proto.req_size], std.mem.asBytes(&req)); - const slen = @min(send.len, proto.max_payload); - @memcpy(msg[proto.req_size..][0..slen], send[0..slen]); + var message: [protocol.message_maximum]u8 = undefined; + @memcpy(message[0..protocol.req_size], std.mem.asBytes(&req)); + const slen = @min(send.len, protocol.maximum_payload); + @memcpy(message[protocol.req_size..][0..slen], send[0..slen]); - var rbuf: [proto.msg_max]u8 = undefined; - const n = ipc.call(h, msg[0 .. proto.req_size + slen], &rbuf) catch return null; - if (n < proto.reply_size) return null; - const reply = std.mem.bytesToValue(proto.Reply, rbuf[0..proto.reply_size]); - const rpl = @min(n - proto.reply_size, out.len); - @memcpy(out[0..rpl], rbuf[proto.reply_size..][0..rpl]); + var rbuf: [protocol.message_maximum]u8 = undefined; + const n = ipc.call(h, message[0 .. protocol.req_size + slen], &rbuf) catch return null; + if (n < protocol.reply_size) return null; + const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]); + const rpl = @min(n - protocol.reply_size, out.len); + @memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]); return .{ .reply = reply, .payload = out[0..rpl] }; } /// Open (or create, with O_CREAT) `path`; returns an fd or -1. pub fn open(path: []const u8, flags: u32) i32 { const fd = allocFd() orelse return -1; - const req = proto.Request{ .op = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags }; + const req = protocol.Request{ .op = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags }; const r = transact(req, path, &.{}) orelse { fds[fd].used = false; return -1; @@ -75,17 +75,17 @@ pub fn open(path: []const u8, flags: u32) i32 { } fn fdPtr(fd: i32) ?*Fd { - if (fd < 0 or fd >= max_fds) return null; + if (fd < 0 or fd >= maximum_fds) return null; const f = &fds[@intCast(fd)]; return if (f.used) f else null; } -/// Read up to `buf.len` bytes at the current offset; returns the count or -1. -pub fn read(fd: i32, buf: []u8) isize { +/// Read up to `buffer.len` bytes at the current offset; returns the count or -1. +pub fn read(fd: i32, buffer: []u8) isize { const f = fdPtr(fd) orelse return -1; - const want: u32 = @intCast(@min(buf.len, proto.max_payload)); - const req = proto.Request{ .op = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 }; - const r = transact(req, &.{}, buf) orelse return -1; + const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload)); + const req = protocol.Request{ .op = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 }; + const r = transact(req, &.{}, buffer) orelse return -1; if (r.reply.status != 0) return -1; f.offset += r.reply.len; return @intCast(r.reply.len); @@ -94,8 +94,8 @@ pub fn read(fd: i32, buf: []u8) isize { /// Write `data` at the current offset; returns the count or -1. pub fn write(fd: i32, data: []const u8) isize { const f = fdPtr(fd) orelse return -1; - const want: u32 = @intCast(@min(data.len, proto.max_payload)); - const req = proto.Request{ .op = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 }; + const want: u32 = @intCast(@min(data.len, protocol.maximum_payload)); + const req = protocol.Request{ .op = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 }; const r = transact(req, data[0..want], &.{}) orelse return -1; if (r.reply.status != 0) return -1; f.offset += r.reply.len; @@ -108,13 +108,13 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 { const f = fdPtr(fd) orelse return -1; const base: i64 = switch (whence) { SEEK_SET => 0, - SEEK_CUR => @intCast(f.offset), + SEEK_CURRENT => @intCast(f.offset), SEEK_END => blk: { - const req = proto.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; - var sbuf: [@sizeOf(proto.Stat)]u8 = undefined; + const req = protocol.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; + var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined; const r = transact(req, &.{}, &sbuf) orelse return -1; - if (r.reply.status != 0 or r.payload.len < @sizeOf(proto.Stat)) return -1; - const st = std.mem.bytesToValue(proto.Stat, sbuf[0..@sizeOf(proto.Stat)]); + if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1; + const st = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]); break :blk @intCast(st.size); }, else => return -1, @@ -126,24 +126,24 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 { } /// Stat `path`. Returns 0 or -1. -pub fn stat(path: []const u8, out: *proto.Stat) i32 { +pub fn stat(path: []const u8, out: *protocol.Stat) i32 { // Open, stat by node, close — simple and enough for now. const fd = open(path, 0); if (fd < 0) return -1; defer close(fd); const f = fdPtr(fd).?; - const req = proto.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; - var sbuf: [@sizeOf(proto.Stat)]u8 = undefined; + const req = protocol.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; + var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined; const r = transact(req, &.{}, &sbuf) orelse return -1; - if (r.reply.status != 0 or r.payload.len < @sizeOf(proto.Stat)) return -1; - out.* = std.mem.bytesToValue(proto.Stat, sbuf[0..@sizeOf(proto.Stat)]); + if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1; + out.* = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]); return 0; } /// Close an fd (best effort — tells the VFS to release the open file). pub fn close(fd: i32) void { const f = fdPtr(fd) orelse return; - const req = proto.Request{ .op = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; + const req = protocol.Request{ .op = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; _ = transact(req, &.{}, &.{}); f.used = false; } diff --git a/lib/vfs_proto.zig b/lib/vfs-protocol.zig similarity index 85% rename from lib/vfs_proto.zig rename to lib/vfs-protocol.zig index 8ba33de..8d71382 100644 --- a/lib/vfs_proto.zig +++ b/lib/vfs-protocol.zig @@ -1,8 +1,8 @@ //! The VFS wire protocol — the message format spoken between a client (via the -//! `rt` file API) and the user-space VFS server over IPC. A request is a fixed +//! `runtime` file API) and the user-space VFS server over IPC. A request is a fixed //! `Request` header followed by an inline payload (a path, or write bytes); a //! reply is a fixed `Reply` header followed by an inline payload (read bytes, or -//! a Stat). Everything fits in one IPC message (<= ipc MSG_MAX = 256 bytes). +//! a Stat). Everything fits in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes). //! //! This is user-space only — the kernel knows nothing of files or paths; it only //! moves the bytes. Shared by lib/unistd.zig (client) and sbin/vfs.zig (server). @@ -43,11 +43,11 @@ pub const Stat = extern struct { _pad: u32 = 0, }; -pub const msg_max: usize = 256; +pub const message_maximum: usize = 256; pub const req_size: usize = @sizeOf(Request); pub const reply_size: usize = @sizeOf(Reply); /// Largest inline payload that still fits one IPC message alongside a header. -pub const max_payload: usize = msg_max - req_size; +pub const maximum_payload: usize = message_maximum - req_size; /// Open flags. pub const O_CREAT: u32 = 1; diff --git a/sbin/busd.zig b/sbin/busd.zig new file mode 100644 index 0000000..6274b68 --- /dev/null +++ b/sbin/busd.zig @@ -0,0 +1,210 @@ +//! /sbin/busd — a user-space **bus driver**, and the smallest honest example of one. +//! +//! A bus driver owns a device that *contains other devices*, enumerates them by some +//! bus-specific protocol, and publishes each one into the kernel's device table so a +//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here +//! the "bus" is the HPET's register block and the "devices" are its comparators, each +//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently. +//! +//! It's a toy bus, but nothing about the mechanism is: `busd` reads how many children +//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per +//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks +//! every one of those resources is contained in what `busd` was granted. A comparator +//! driver then claims a child and maps only *its* registers — not the whole block. +//! +//! It also proves the negative: registering a child whose window escapes the parent's +//! is refused. Without that check, `device_register` would be a system_call for mapping +//! arbitrary physical memory. + +const std = @import("std"); +const runtime = @import("runtime"); +const device = runtime.device; + +const register_general_cap = 0x000; + +/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes. +fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor { + return .{ + .kind = @intFromEnum(device.ResourceKind.memory), + .start = hpet_base + 0x100 + 0x20 * n, + .len = 0x20, + }; +} + +fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor { + const total = device.enumerate(buffer); + const n = @min(total, buffer.len); + for (buffer[0..n]) |d| { + if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; + if (d.parent != device.no_parent) continue; // the block, not a comparator child + for (0..d.resource_count) |j| { + if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d; + } + } + return null; +} + +/// The parent's MMIO resource, and its IRQ if it has one. +fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } { + var mmio: device.ResourceDescriptor = undefined; + var irq: ?device.ResourceDescriptor = null; + for (0..d.resource_count) |j| { + const r = d.resources[j]; + if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r; + if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r; + } + return .{ .mmio = mmio, .irq = irq }; +} + +fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 { + for (buffer[0..@min(total, buffer.len)]) |d| { + if (d.parent == parent_id) return d.id; + } + return null; +} + +pub fn main() void { + const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch { + _ = runtime.system.write("busd: out of memory\n"); + return; + }; + + const parent = findHpet(buffer) orelse { + _ = runtime.system.write("busd: no HPET\n"); + return; + }; + const resource = resourcesOf(parent); + + // Claim the bus. Everything below is subdivision of what this claim granted. + // + // Claims are exclusive, and at a normal boot the kernel spawns every initrd + // binary — so hpetd may own the HPET already. That's not an error, it's the + // capability model working: exit quietly and leave the device to its owner. The + // `bus` test spawns busd alone, so there it wins the claim. + if (!device.claim(parent.id)) { + _ = runtime.system.write("busd: HPET already claimed by another driver, nothing to do\n"); + return; + } + + // Enumerate the bus: ask the hardware how many children it has. + const base = device.mmioMap(parent.id, 0) orelse { + _ = runtime.system.write("busd: mmio_map failed\n"); + return; + }; + const cap: *volatile u64 = @ptrFromInt(base + register_general_cap); + const n_children = ((cap.* >> 8) & 0x1F) + 1; + + // Publish one child per comparator, each owning only its own window. + var published: u64 = 0; + var n: u64 = 0; + while (n < n_children) : (n += 1) { + var child = std.mem.zeroes(device.DeviceDescriptor); + child.class = @intFromEnum(device.DeviceClass.timer); + child.hid_len = 6; + child.hid[0..6].* = "hpet-t".*; + child.resource_count = 1; + child.resources[0] = timerWindow(resource.mmio.start, n); + // Comparators share the block's interrupt line; only one child can bind it, + // but all of them may legitimately name it. + if (resource.irq) |i| { + child.resources[child.resource_count] = i; + child.resource_count += 1; + } + + if (device.register(parent.id, &child) == null) { + _ = runtime.system.write("busd: register failed\n"); + return; + } + published += 1; + } + + // The negative case. A window one byte past the end of the parent's must be + // refused — otherwise device_register would be "map any physical page you like". + // Confirm the table did not grow, not merely that the call returned null: null + // also means NoSpace/BadParent, so a size check is what actually proves the + // *containment* rule fired. + const before = device.enumerate(buffer); + var rogue = std.mem.zeroes(device.DeviceDescriptor); + rogue.class = @intFromEnum(device.DeviceClass.unknown); + rogue.resource_count = 1; + rogue.resources[0] = .{ + .kind = @intFromEnum(device.ResourceKind.memory), + .start = resource.mmio.start + resource.mmio.len, + .len = 0x1000, + }; + if (device.register(parent.id, &rogue) != null) { + _ = runtime.system.write("busd: FAIL out-of-window child was accepted\n"); + return; + } + if (device.enumerate(buffer) != before) { + _ = runtime.system.write("busd: FAIL rogue child leaked into the table\n"); + return; + } + + // And confirm the children came back with the right parent and a *narrower* + // window than the bus — read from the table, not from our own memory. + const total = device.enumerate(buffer); + var seen: u64 = 0; + for (buffer[0..@min(total, buffer.len)]) |d| { + if (d.parent != parent.id) continue; + const w = d.resources[0]; + if (w.start < resource.mmio.start or w.len >= resource.mmio.len) { + _ = runtime.system.write("busd: FAIL child window is not inside the bus\n"); + return; + } + seen += 1; + } + if (seen != published) { + _ = runtime.system.write("busd: FAIL child count mismatch\n"); + return; + } + + // Delegation, end to end: claim a child and map *it*. A real class driver would be + // a different process; here busd plays both parts, which exercises the same path. + // The child's window is 0x20 bytes at parent+0x100, so the register it sees at + // offset 0 must be the same timer-0 configuration register the bus sees at 0x100. + // + // (mmio_map rounds to a page, so the child's mapping physically covers the whole + // 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md. + // The *resource* is narrow even though the page isn't.) + const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse { + _ = runtime.system.write("busd: FAIL no child to claim\n"); + return; + }; + if (!device.claim(child_id)) { + _ = runtime.system.write("busd: FAIL could not claim own child\n"); + return; + } + const child_base = device.mmioMap(child_id, 0) orelse { + _ = runtime.system.write("busd: FAIL child mmio_map refused\n"); + return; + }; + const via_child: *volatile u64 = @ptrFromInt(child_base); + const via_bus: *volatile u64 = @ptrFromInt(base + 0x100); + if (via_child.* != via_bus.*) { + _ = runtime.system.write("busd: FAIL child window does not alias the bus register\n"); + return; + } + + // A descriptor pointer into an unmapped page must fail the call, not fault the + // kernel. Grab a page, free it, and register through the stale address: if the + // kernel dereferenced it raw (rather than copying in through the page tables) this + // would triple-fault QEMU and the test would time out instead of printing ok. + const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE); + if (!runtime.system.mmapFailed(scratch)) { + _ = runtime.system.munmap(scratch, 0x1000); + const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch); + if (device.register(parent.id, descriptor) != null) { + _ = runtime.system.write("busd: FAIL register accepted an unmapped descriptor\n"); + return; + } + } + + _ = runtime.system.write("busd: ok\n"); + while (true) runtime.system.sleep(1000); +} + +pub const panic = runtime.panic; +comptime { + _ = &runtime.start._start; +} diff --git a/sbin/hpetd.zig b/sbin/hpetd.zig index 2ff1d96..dadf344 100644 --- a/sbin/hpetd.zig +++ b/sbin/hpetd.zig @@ -1,75 +1,187 @@ -//! /sbin/hpetd — a user-space HPET driver, the first real device driver. It -//! proves IO passthrough end to end: enumerate the device table, find the HPET -//! (a timer with an MMIO window), claim it, map its registers directly into this -//! ring-3 address space (strong-uncacheable), then drive the hardware — enable -//! the main counter and read it. If the counter advances, a user process is -//! touching real hardware through a kernel-granted MMIO mapping. +//! /sbin/hpetd — a user-space HPET driver. It proves the whole driver model end to +//! end: enumerate the device table, find the HPET, claim it, map its registers into +//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC +//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it. //! -//! Register offsets (HPET spec): general config = 0x10 (bit 0 = ENABLE), -//! main counter = 0xF0. +//! Nothing here polls. Between interrupts the process is `.blocked` and off every +//! scheduler queue; the core runs other work or idles. That is the point of the +//! exercise — a driver is a process that sleeps until its device has something to +//! say (see docs/drivers.md). +//! +//! The comparator is configured **level-triggered** on purpose. Edge would be +//! simpler, but level is the discipline every real device line needs, and it forces +//! the full cycle to be correct: +//! +//! kernel ISR mask the GSI -> EOI -> notify this endpoint +//! hpetd wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm +//! hpetd irq_ack -> kernel unmasks the GSI +//! +//! Clear the status bit *before* acking, or the line is still asserted when the +//! kernel unmasks and the I/O APIC redelivers forever. +//! +//! Register map (HPET spec 1.0a): +//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1 +//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF +//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear) +//! 0x0F0 MAIN_COUNTER +//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic) +//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP +//! 0x108 TIMER0_COMPARATOR -const rt = @import("rt"); -const dev = rt.dev; +const runtime = @import("runtime"); +const device = runtime.device; +const ipc = runtime.ipc; + +const register_general_cap = 0x000; +const register_general_configuration = 0x010; +const register_int_status = 0x020; +const register_main_counter = 0x0F0; +const register_timer0_configuration = 0x100; +const register_timer0_comparator = 0x108; + +const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF +const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF +const tn_int_type_level: u64 = 1 << 1; +const tn_int_enb: u64 = 1 << 2; +const tn_type_periodic: u64 = 1 << 3; +const tn_route_shift = 9; +const tn_route_mask: u64 = 0x1F << tn_route_shift; + +/// Interrupts to observe before declaring victory. +const target_ticks = 5; + +fn register(base: usize, off: usize) *volatile u64 { + return @ptrFromInt(base + off); +} + +/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two +/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`. +const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 }; + +fn findHpet(buffer: []device.DeviceDescriptor) ?Found { + const total = device.enumerate(buffer); + const n = @min(total, buffer.len); + for (buffer[0..n]) |d| { + if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; + // Skip comparator children a bus driver may have published below the block + // (see sbin/busd.zig) — we want the register block itself. + if (d.parent != device.no_parent) continue; + var mmio: ?u64 = null; + var irq: ?u64 = null; + for (0..d.resource_count) |j| { + switch (d.resources[j].kind) { + @intFromEnum(device.ResourceKind.memory) => mmio = mmio orelse j, + @intFromEnum(device.ResourceKind.irq) => irq = irq orelse j, + else => {}, + } + } + if (mmio) |m| if (irq) |i| { + return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start }; + }; + } + return null; +} pub fn main() void { // Enumerate into a heap buffer (too big for the one-page user stack). - const buf = rt.allocator().alloc(dev.DeviceDesc, 32) catch { - _ = rt.sys.write("hpetd: out of memory\n"); - return; - }; - const total = dev.enumerate(buf); - const n = @min(total, buf.len); - - // Find a timer-class device with an MMIO resource (the HPET). - var dev_id: u64 = 0; - var res_idx: u64 = 0; - var found = false; - var i: usize = 0; - outer: while (i < n) : (i += 1) { - const d = buf[i]; - if (d.class != @intFromEnum(dev.DeviceClass.timer)) continue; - var j: usize = 0; - while (j < d.resource_count) : (j += 1) { - if (d.resources[j].kind == @intFromEnum(dev.ResourceKind.memory)) { - dev_id = d.id; - res_idx = j; - found = true; - break :outer; - } - } - } - if (!found) { - _ = rt.sys.write("hpetd: no HPET found\n"); - return; - } - - if (!dev.claim(dev_id)) { - _ = rt.sys.write("hpetd: claim failed\n"); - return; - } - const base = dev.mmioMap(dev_id, res_idx) orelse { - _ = rt.sys.write("hpetd: mmio_map failed\n"); + const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch { + _ = runtime.system.write("hpetd: out of memory\n"); return; }; - // Drive the hardware: enable the counter (an MMIO write), then read it twice. - const config: *volatile u64 = @ptrFromInt(base + 0x10); - config.* |= 1; // ENABLE - const counter: *volatile u64 = @ptrFromInt(base + 0xF0); - const a = counter.*; - rt.sys.sleep(50); - const b = counter.*; + const hpet = findHpet(buffer) orelse { + _ = runtime.system.write("hpetd: no HPET with an IRQ\n"); + return; + }; - if (b > a) { - while (true) { - _ = rt.sys.write("hpetd: ok\n"); - rt.sys.sleep(1000); + if (!device.claim(hpet.device_id)) { + _ = runtime.system.write("hpetd: claim failed\n"); + return; + } + const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse { + _ = runtime.system.write("hpetd: mmio_map failed\n"); + return; + }; + + // The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator + // to raise exactly this line — the kernel will only bind the one it recorded. + const gsi = hpet.gsi; + + const endpoint = ipc.createEndpoint() orelse { + _ = runtime.system.write("hpetd: create_endpoint failed\n"); + return; + }; + + // --- program the hardware ------------------------------------------------ + // Counter period, so we can arm the comparator a fixed wall-clock distance out. + const femtos_per_tick = register(base, register_general_cap).* >> 32; + if (femtos_per_tick == 0) { + _ = runtime.system.write("hpetd: bad HPET period\n"); + return; + } + const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick; + + // Stop the counter and take the legacy route off while we reconfigure. + register(base, register_general_configuration).* &= ~(configuration_enable | configuration_leg_rt); + + // Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled. + // One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk — + // we simply re-arm from the driver on each interrupt, which is what a tickless + // timer driver does anyway. + var t0 = register(base, register_timer0_configuration).*; + t0 &= ~(tn_route_mask | tn_type_periodic); + t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift); + register(base, register_timer0_configuration).* = t0; + + // Clear any stale assertion, then arm ~100 ms out and start the counter. + register(base, register_int_status).* = 1; + register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100; + register(base, register_general_configuration).* |= configuration_enable; + + if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) { + _ = runtime.system.write("hpetd: irq_bind failed\n"); + return; + } + _ = runtime.system.write("hpetd: bound, sleeping until the hardware speaks\n"); + + // --- the driver loop ----------------------------------------------------- + // Blocked in replyWait. No polling, no spinning: the next line of this function + // runs only because an interrupt fired. + var receive: [64]u8 = undefined; + var count: usize = 0; + while (count < target_ticks) { + // Blocked here. The task is `.blocked` and off every scheduler queue; the + // next line runs only because the HPET raised its line. + const r = ipc.replyWait(endpoint, &.{}, &receive); + if (!r.isNotification()) continue; // a client request, not our IRQ + + // Quiet the device: write 1 to timer 0's status bit. Until this lands, the + // line is still asserted and unmasking would refire immediately. + register(base, register_int_status).* = 1; + count += 1; + + if (count < target_ticks) { + register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100; + } else { + // Last one: stop the source rather than re-arming, so the line is left + // both quiet *and* unmasked by the ack below. Re-arming here would leave + // a pending interrupt that nobody is waiting for, and the ISR would mask + // the line again a moment later. + register(base, register_timer0_configuration).* &= ~tn_int_enb; + } + + _ = runtime.system.write("hpetd: irq\n"); + if (!device.irqAck(hpet.device_id, hpet.irq)) { + _ = runtime.system.write("hpetd: irq_ack failed\n"); + return; } } - _ = rt.sys.write("hpetd: counter stuck\n"); + + _ = runtime.system.write("hpetd: ok\n"); + while (true) runtime.system.sleep(1000); } -pub const panic = rt.panic; +pub const panic = runtime.panic; comptime { - _ = &rt.start._start; + _ = &runtime.start._start; } diff --git a/sbin/init.zig b/sbin/init.zig index c4598e7..f75bda5 100644 --- a/sbin/init.zig +++ b/sbin/init.zig @@ -2,7 +2,7 @@ //! freestanding binary (see build.zig), shipped on the boot volume at sbin/init, //! loaded by the bootloader, and started in ring 3 as a scheduled process by the //! kernel (src/kernel/process.zig). It links against the shared user runtime -//! library `rt` and talks to the kernel only through `rt`'s syscall wrappers. +//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers. //! //! Today it proves the C-convention heap works, then settles into a heartbeat: //! it prints a line and sleeps, forever — enough to show the system reaches user @@ -10,7 +10,7 @@ //! idle loop. It grows into the real init (service supervision) once there are //! other user programs to supervise. -const rt = @import("rt"); +const runtime = @import("runtime"); pub fn main() void { // Prove the heap end to end: allocate through the runtime allocator (which @@ -19,21 +19,21 @@ pub fn main() void { // free it. A fault here would kill init before it heartbeats — so the init // test doubles as the heap regression test. (C code links the same heap via // the extern malloc/free symbols; Zig code uses this allocator.) - const gpa = rt.allocator(); - if (gpa.alloc(u8, 64)) |buf| { - const msg = "init: heap ok\n"; - @memcpy(buf[0..msg.len], msg); - _ = rt.sys.write(buf[0..msg.len]); - gpa.free(buf); + const gpa = runtime.allocator(); + if (gpa.alloc(u8, 64)) |buffer| { + const message = "init: heap ok\n"; + @memcpy(buffer[0..message.len], message); + _ = runtime.system.write(buffer[0..message.len]); + gpa.free(buffer); } else |_| {} while (true) { - _ = rt.sys.write("init: heartbeat\n"); - rt.sys.sleep(1000); + _ = runtime.system.write("init: heartbeat\n"); + runtime.system.sleep(1000); } } -pub const panic = rt.panic; +pub const panic = runtime.panic; comptime { - _ = &rt.start._start; // pull the runtime entry shim into the image + _ = &runtime.start._start; // pull the runtime entry shim into the image } diff --git a/sbin/vfstest.zig b/sbin/vfs-test.zig similarity index 52% rename from sbin/vfstest.zig rename to sbin/vfs-test.zig index b8da8ed..00cabfe 100644 --- a/sbin/vfstest.zig +++ b/sbin/vfs-test.zig @@ -1,13 +1,13 @@ //! /sbin/vfstest — a client that proves the VFS round trip end to end: open a -//! file through the `rt` file API, write to it, seek back, read it, and compare. +//! file through the `runtime` file API, write to it, seek back, read it, and compare. //! On success it heartbeats "vfstest: ok" so the kernel test can observe it; //! on failure it reports what went wrong. Shipped in the initrd alongside vfs. const std = @import("std"); -const rt = @import("rt"); +const runtime = @import("runtime"); pub fn main() void { - const u = rt.unistd; + const u = runtime.unistd; const payload = "hello-vfs"; // The VFS server may not have registered yet — retry open until it's up. @@ -15,33 +15,33 @@ pub fn main() void { var tries: u32 = 0; while (fd < 0 and tries < 200) : (tries += 1) { fd = u.open("greeting", u.O_CREAT); - if (fd < 0) rt.sys.sleep(20); + if (fd < 0) runtime.system.sleep(20); } if (fd < 0) { - _ = rt.sys.write("vfstest: open failed\n"); + _ = runtime.system.write("vfstest: open failed\n"); return; } if (u.write(fd, payload) != @as(isize, payload.len)) { - _ = rt.sys.write("vfstest: write failed\n"); + _ = runtime.system.write("vfstest: write failed\n"); return; } _ = u.lseek(fd, 0, u.SEEK_SET); - var buf: [32]u8 = undefined; - const n = u.read(fd, &buf); + var buffer: [32]u8 = undefined; + const n = u.read(fd, &buffer); u.close(fd); - if (n == @as(isize, payload.len) and std.mem.eql(u8, buf[0..@intCast(n)], payload)) { + if (n == @as(isize, payload.len) and std.mem.eql(u8, buffer[0..@intCast(n)], payload)) { while (true) { - _ = rt.sys.write("vfstest: ok\n"); - rt.sys.sleep(1000); + _ = runtime.system.write("vfstest: ok\n"); + runtime.system.sleep(1000); } } - _ = rt.sys.write("vfstest: mismatch\n"); + _ = runtime.system.write("vfstest: mismatch\n"); } -pub const panic = rt.panic; +pub const panic = runtime.panic; comptime { - _ = &rt.start._start; + _ = &runtime.start._start; } diff --git a/sbin/vfs.zig b/sbin/vfs.zig index 9222dca..fc407e1 100644 --- a/sbin/vfs.zig +++ b/sbin/vfs.zig @@ -1,16 +1,16 @@ //! /sbin/vfs — the user-space VFS server. Shipped in the initrd, spawned as a -//! ring-3 process, and reached by every other process through IPC (the `rt` +//! ring-3 process, and reached by every other process through IPC (the `runtime` //! file API marshals open/read/write/stat/close into calls to this server's //! endpoint, published under the well-known `vfs` service id). //! //! For now the namespace is a small in-memory ramfs (opening a name creates it): //! enough to prove the whole path — client file API -> IPC -> server dispatch -> -//! reply. Device nodes backed by user-space drivers (/dev) layer on top in M10, -//! where `open` on a /dev name forwards to the owning driver's endpoint. +//! reply. Device nodes backed by user-space drivers (/device) layer on top in M10, +//! where `open` on a /device name forwards to the owning driver's endpoint. const std = @import("std"); -const rt = @import("rt"); -const proto = rt.vfsproto; +const runtime = @import("runtime"); +const protocol = runtime.vfs_protocol; const Node = struct { used: bool = false, @@ -54,11 +54,11 @@ fn openAt(id: u64) ?*OpenFile { } /// Serialise a reply header + payload into `out`; returns the total length. -fn writeReply(out: []u8, reply: proto.Reply, payload: []const u8) usize { - @memcpy(out[0..proto.reply_size], std.mem.asBytes(&reply)); - const n = @min(payload.len, out.len - proto.reply_size); - @memcpy(out[proto.reply_size..][0..n], payload[0..n]); - return proto.reply_size + n; +fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize { + @memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply)); + const n = @min(payload.len, out.len - protocol.reply_size); + @memcpy(out[protocol.reply_size..][0..n], payload[0..n]); + return protocol.reply_size + n; } fn fail(out: []u8) usize { @@ -66,10 +66,10 @@ fn fail(out: []u8) usize { } /// Handle one request; write the reply into `out`, return its length. -fn handle(msg: []const u8, out: []u8) usize { - if (msg.len < proto.req_size) return fail(out); - const req = std.mem.bytesToValue(proto.Request, msg[0..proto.req_size]); - const payload = msg[proto.req_size..]; +fn handle(message: []const u8, out: []u8) usize { + if (message.len < protocol.req_size) return fail(out); + const req = std.mem.bytesToValue(protocol.Request, message[0..protocol.req_size]); + const payload = message[protocol.req_size..]; switch (req.op) { .open => { @@ -88,7 +88,7 @@ fn handle(msg: []const u8, out: []u8) usize { const nd = &nodes[of.node]; const off: usize = @intCast(req.offset); if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF - const n = @min(@min(nd.size - off, req.len), proto.max_payload); + const n = @min(@min(nd.size - off, req.len), protocol.maximum_payload); return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]); }, .write => { @@ -103,8 +103,8 @@ fn handle(msg: []const u8, out: []u8) usize { }, .stat => { const of = openAt(req.node) orelse return fail(out); - const st = proto.Stat{ .size = nodes[of.node].size, .kind = 0 }; - return writeReply(out, .{ .status = 0, .len = @sizeOf(proto.Stat) }, std.mem.asBytes(&st)); + const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 }; + return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st)); }, .close => { if (req.node < opens.len) opens[@intCast(req.node)].used = false; @@ -114,27 +114,27 @@ fn handle(msg: []const u8, out: []u8) usize { } pub fn main() void { - const ep = rt.ipc.createEndpoint() orelse { - _ = rt.sys.write("vfs: no endpoint\n"); + const endpoint = runtime.ipc.createEndpoint() orelse { + _ = runtime.system.write("vfs: no endpoint\n"); return; }; - if (!rt.ipc.register(.vfs, ep)) { - _ = rt.sys.write("vfs: register failed\n"); + if (!runtime.ipc.register(.vfs, endpoint)) { + _ = runtime.system.write("vfs: register failed\n"); return; } - _ = rt.sys.write("vfs: ready\n"); + _ = runtime.system.write("vfs: ready\n"); - var reply_buf: [proto.msg_max]u8 = undefined; + var reply_buffer: [protocol.message_maximum]u8 = undefined; var reply_len: usize = 0; - var recv: [proto.msg_max]u8 = undefined; + var receive: [protocol.message_maximum]u8 = undefined; while (true) { - const got = rt.ipc.replyWait(ep, reply_buf[0..reply_len], &recv); + const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive); // Ignore notifications (none expected here); handle a request. - reply_len = handle(recv[0..got.len], &reply_buf); + reply_len = handle(receive[0..got.len], &reply_buffer); } } -pub const panic = rt.panic; +pub const panic = runtime.panic; comptime { - _ = &rt.start._start; + _ = &runtime.start._start; } diff --git a/src/boot/efi.zig b/src/boot/efi.zig index 0a612f5..ee16464 100644 --- a/src/boot/efi.zig +++ b/src/boot/efi.zig @@ -2,7 +2,7 @@ const std = @import("std"); const uefi = std.os.uefi; const elf = std.elf; const danos = @import("danos"); -const BootInfo = danos.BootInfo; +const BootInformation = danos.BootInformation; const GraphicsOutput = uefi.protocol.GraphicsOutput; const EdidActive = uefi.protocol.edid.Active; const MemoryMapSlice = uefi.tables.MemoryMapSlice; @@ -40,7 +40,7 @@ fn boot() !noreturn { // Everything the kernel needs must be gathered *before* we exit boot // services, since afterwards none of these calls are usable. - var boot_info: BootInfo = .{ + var boot_information: BootInformation = .{ // A missing GOP (a headless machine) is not fatal — hand the kernel a // "no framebuffer" descriptor (base 0) and let it log to serial instead. .framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{ @@ -59,17 +59,17 @@ fn boot() !noreturn { .acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0, }; - const entry = try loadKernel(bs, &boot_info); + const entry = try loadKernel(bs, &boot_information); // Best effort: a volume without sbin/init still boots (kernel-only). - loadInit(bs, &boot_info) catch |err| { + loadInit(bs, &boot_information) catch |err| { log("danos: no sbin/init ("); logBytes(@errorName(err)); log(") - booting without user space\r\n"); }; // Best effort: the initrd (VFS server + drivers) is optional too. - loadInitrd(bs, &boot_info) catch |err| { + loadInitrd(bs, &boot_information) catch |err| { log("danos: no initrd ("); logBytes(@errorName(err)); log(")\r\n"); @@ -80,17 +80,17 @@ fn boot() !noreturn { // now, while boot services (and the memory map) are still stable — nothing // is allocatable after ExitBootServices, and any allocation between fetching // the map and exiting would invalidate the map key. - const cr3 = try buildBootstrapTables(bs, &boot_info); + const cr3 = try buildBootstrapTables(bs, &boot_information); log("danos: kernel loaded, exiting boot services\r\n"); - boot_info.memory_map = try exitBootServices(bs); + boot_information.memory_map = try exitBootServices(bs); // Switch onto our tables and jump to the kernel in one uninterruptible step. - // We load RDI explicitly (SysV first arg) rather than trusting this UEFI + // We load RDI explicitly (SystemV first arg) rather than trusting this UEFI // binary's Microsoft-x64 default, and jump straight to the (possibly // higher-half) entry — the bootstrap tables map both the low loader code // executing this and the kernel's link address. - handoff(cr3, entry, &boot_info); + handoff(cr3, entry, &boot_information); } /// A display resolution in pixels. @@ -195,7 +195,7 @@ fn edidNative(edid: []const u8) ?Resolution { /// Open the kernel on the volume we booted from, read it into a pool buffer, /// load its segments, and return the physical entry-point address. -fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize { +fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize { const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse return error.NoLoadedImage; const device = loaded.device_handle orelse return error.NoBootDevice; @@ -224,7 +224,7 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize { read_total += n; } - return loadElf(bs, image, boot_info); + return loadElf(bs, image, boot_information); } // --- bootstrap page tables ------------------------------------------------- @@ -241,7 +241,7 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize { const pte_present: u64 = 1 << 0; const pte_write: u64 = 1 << 1; const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level -const pte_addr: u64 = 0x000F_FFFF_FFFF_F000; +const pte_address: u64 = 0x000F_FFFF_FFFF_F000; const gib: u64 = 1 << 30; /// A bump allocator over a pre-reserved block of zeroed frames, for page tables. @@ -258,40 +258,40 @@ const TablePool = struct { return frame; } - fn table(phys: u64) *[512]u64 { - return @ptrFromInt(phys); + fn table(physical: u64) *[512]u64 { + return @ptrFromInt(physical); } /// Return the next-level table an entry points at, creating it if absent. fn descend(self: *TablePool, entry: *u64) !u64 { - if (entry.* & pte_present != 0) return entry.* & pte_addr; + if (entry.* & pte_present != 0) return entry.* & pte_address; const frame = try self.alloc(); entry.* = frame | pte_present | pte_write; return frame; } - fn map2M(self: *TablePool, pml4: u64, virt: u64, phys: u64) !void { - const pml4e = &table(pml4)[(virt >> 39) & 0x1FF]; + fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void { + const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF]; const pdpt = try self.descend(pml4e); - const pdpte = &table(pdpt)[(virt >> 30) & 0x1FF]; + const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF]; const pd = try self.descend(pdpte); - table(pd)[(virt >> 21) & 0x1FF] = (phys & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps; + table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps; } - fn map4K(self: *TablePool, pml4: u64, virt: u64, phys: u64) !void { - const pml4e = &table(pml4)[(virt >> 39) & 0x1FF]; + fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void { + const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF]; const pdpt = try self.descend(pml4e); - const pdpte = &table(pdpt)[(virt >> 30) & 0x1FF]; + const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF]; const pd = try self.descend(pdpte); - const pde = &table(pd)[(virt >> 21) & 0x1FF]; + const pde = &table(pd)[(virtual >> 21) & 0x1FF]; const pt = try self.descend(pde); - table(pt)[(virt >> 12) & 0x1FF] = (phys & pte_addr) | pte_present | pte_write; + table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write; } }; /// Build the bootstrap tables and return the physical PML4 address (for CR3). /// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant. -fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_info: *const BootInfo) !u64 { +fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 { // 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for // the 4 GiB identity+physmap ranges, plus the kernel image's PTs. const pool_pages = 64; @@ -303,44 +303,44 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_info: *const BootInf // Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO // (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a // framebuffer above 4 GiB would extend this — see the fb window below. - var addr: u64 = 0; - while (addr < 4 * gib) : (addr += 2 << 20) { - try pool.map2M(pml4, addr, addr); // identity - try pool.map2M(pml4, danos.physToVirt(addr), addr); // physmap + var address: u64 = 0; + while (address < 4 * gib) : (address += 2 << 20) { + try pool.map2M(pml4, address, address); // identity + try pool.map2M(pml4, danos.physicalToVirtual(address), address); // physmap } // A framebuffer above the 4 GiB window needs its own identity + physmap // pages (the kernel touches fb.base before it builds its own tables). - const fb = boot_info.framebuffer; + const fb = boot_information.framebuffer; if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) { var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF); const fb_end = fb.base + @as(u64, fb.pitch) * fb.height; while (p < fb_end) : (p += 2 << 20) { try pool.map2M(pml4, p, p); - try pool.map2M(pml4, danos.physToVirt(p), p); + try pool.map2M(pml4, danos.physicalToVirtual(p), p); } } - // Higher-half kernel segments (virt != phys). While the kernel still links + // Higher-half kernel segments (virtual != physical). While the kernel still links // low its segments sit in the identity range and need no separate mapping // (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so // only map segments that actually live in the higher half. - for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| { - if (seg.virt < danos.kernel_virt_base) continue; + for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| { + if (seg.virtual < danos.kernel_virt_base) continue; var off: u64 = 0; while (off < seg.pages * page_size) : (off += page_size) { - try pool.map4K(pml4, seg.virt + off, seg.phys + off); + try pool.map4K(pml4, seg.virtual + off, seg.physical + off); } } return pml4; } -/// Switch onto `cr3` and jump to the kernel `entry` with `boot_info` in RDI, +/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI, /// interrupts off, in one block so nothing runs between the CR3 load and the /// jump. The identity mapping keeps this low loader code valid across the CR3 /// load; the jump target is mapped (identity while low, higher-half once high). -fn handoff(cr3: u64, entry: usize, boot_info: *const BootInfo) noreturn { +fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn { asm volatile ( \\cli \\movq %[cr3], %%cr3 @@ -348,7 +348,7 @@ fn handoff(cr3: u64, entry: usize, boot_info: *const BootInfo) noreturn { \\callq *%[entry] : : [cr3] "r" (cr3), - [bi] "r" (boot_info), + [bi] "r" (boot_information), [entry] "r" (entry), : .{ .memory = true }); unreachable; @@ -389,25 +389,25 @@ fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 { /// Ferry the init program (sbin/init) to the kernel. The kernel does the ELF /// loading itself (into ring-3 mappings) — the loader just carries the bytes. -fn loadInit(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void { +fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void { const image = try loadFile(bs, init_file_name); - boot_info.init_base = @intFromPtr(image.ptr); - boot_info.init_len = image.len; + boot_information.init_base = @intFromPtr(image.ptr); + boot_information.init_len = image.len; log("danos: sbin/init loaded\r\n"); } /// Ferry the initrd (the VFS server + drivers) to the kernel, same as init. -fn loadInitrd(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void { +fn loadInitrd(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void { const image = try loadFile(bs, initrd_file_name); - boot_info.initrd_base = @intFromPtr(image.ptr); - boot_info.initrd_len = image.len; + boot_information.initrd_base = @intFromPtr(image.ptr); + boot_information.initrd_len = image.len; log("danos: initrd loaded\r\n"); } /// Validate the ELF, copy every PT_LOAD segment to its physical address, and /// record each segment's layout so the kernel can re-map itself with the right /// permissions. -fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize { +fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize { if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf; const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr)); @@ -442,15 +442,15 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us // Record the virtual link address and the physical load address so the // kernel can map itself with the right permissions post-switch. They're // equal while the kernel links low; they diverge once it links high. - const n = boot_info.kernel_segment_count; - if (n < boot_info.kernel_segments.len) { - boot_info.kernel_segments[n] = .{ - .virt = phdr.p_vaddr, - .phys = phdr.p_paddr, + const n = boot_information.kernel_segment_count; + if (n < boot_information.kernel_segments.len) { + boot_information.kernel_segments[n] = .{ + .virtual = phdr.p_vaddr, + .physical = phdr.p_paddr, .pages = pages, .flags = phdr.p_flags, }; - boot_info.kernel_segment_count = n + 1; + boot_information.kernel_segment_count = n + 1; } } @@ -467,21 +467,21 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap { const info = try bs.getMemoryMapInfo(); // Spare descriptors to absorb the growth from the allocations below. const cap = info.len + 8; - const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size); - const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion)); - const map = bs.getMemoryMap(map_buf) catch { - _ = bs.freePool(map_buf.ptr) catch {}; - _ = bs.freePool(regions_buf.ptr) catch {}; + const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size); + const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion)); + const map = bs.getMemoryMap(map_buffer) catch { + _ = bs.freePool(map_buffer.ptr) catch {}; + _ = bs.freePool(regions_buffer.ptr) catch {}; continue; }; bs.exitBootServices(uefi.handle, map.info.key) catch { - _ = bs.freePool(map_buf.ptr) catch {}; - _ = bs.freePool(regions_buf.ptr) catch {}; + _ = bs.freePool(map_buffer.ptr) catch {}; + _ = bs.freePool(regions_buffer.ptr) catch {}; continue; }; // Boot services are gone; do not touch `bs` again. Converting the map is // pure computation on memory we already hold, so it's safe here. - return convertMemoryMap(map, regions_buf); + return convertMemoryMap(map, regions_buffer); } return error.ExitBootServicesFailed; } @@ -513,11 +513,11 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap { // Coalesce with the previous region if it's the same kind and contiguous. if (count > 0) { - const prev = ®ions[count - 1]; - if (prev.kind == kind and - prev.base + prev.pages * danos.page_size == d.physical_start) + const previous = ®ions[count - 1]; + if (previous.kind == kind and + previous.base + previous.pages * danos.page_size == d.physical_start) { - prev.pages += d.number_of_pages; + previous.pages += d.number_of_pages; continue; } } @@ -533,7 +533,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap { /// Map a UEFI descriptor to danos's neutral kind. A region that isn't /// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or -/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio` +/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio` /// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM /// and such holes, and the cache attribute is what actually tells them apart. /// @@ -554,33 +554,33 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind { } /// Write a compile-time string to the console (best effort). -fn log(comptime msg: []const u8) void { +fn log(comptime message: []const u8) void { const out = uefi.system_table.con_out orelse return; - _ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {}; + _ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {}; } /// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16. fn logBytes(bytes: []const u8) void { const out = uefi.system_table.con_out orelse return; - var buf: [128]u16 = undefined; + var buffer: [128]u16 = undefined; var i: usize = 0; for (bytes) |b| { - if (i + 1 >= buf.len) break; - buf[i] = b; + if (i + 1 >= buffer.len) break; + buffer[i] = b; i += 1; } - buf[i] = 0; - _ = out.outputString(buf[0..i :0].ptr) catch {}; + buffer[i] = 0; + _ = out.outputString(buffer[0..i :0].ptr) catch {}; } fn acpiRootSystemDescriptorPointer() ?*const anyopaque { const table_entries = uefi.system_table.number_of_table_entries; - const config_tables = uefi.system_table.configuration_table; + const configuration_tables = uefi.system_table.configuration_table; const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid; const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid; for (0..table_entries) |i| { - const entry = config_tables[i]; + const entry = configuration_tables[i]; if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) { return entry.vendor_table; } diff --git a/src/device/acpi.zig b/src/device/acpi.zig index 827ea1d..c0b9478 100644 --- a/src/device/acpi.zig +++ b/src/device/acpi.zig @@ -11,43 +11,43 @@ //! //! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel //! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO -//! and is *not* mapped up front, so config-space pages are mapped on demand via -//! the `Hal.mapMmio` callback the caller supplies (the arch VMM's map primitive). +//! and is *not* mapped up front, so configuration-space pages are mapped on demand via +//! the `Hal.mapMmio` callback the caller supplies (the architecture VMM's map primitive). const std = @import("std"); const danos = @import("danos"); -const config = @import("config"); -const device = @import("device.zig"); +const parameters = @import("parameters"); +const device_model = @import("device-model.zig"); const aml = @import("aml/aml.zig"); -const DeviceTree = device.DeviceTree; -const Hal = device.Hal; +const DeviceTree = device_model.DeviceTree; +const Hal = device_model.Hal; /// A hardware register located either in MMIO or I/O-port space, as ACPI's /// Generic Address Structure describes. `address == 0` means "not present". -pub const RegAccess = struct { +pub const RegisterAccess = struct { /// true = system memory (MMIO), false = system I/O port space. mmio: bool = false, address: u64 = 0, /// Access width in bytes. width: u8 = 0, - pub fn present(self: RegAccess) bool { + pub fn present(self: RegisterAccess) bool { return self.address != 0; } }; /// Everything the power subsystem needs, extracted from the FADT and the AML /// sleep packages during discovery. Populated by `discover`, read by `power`. -pub const PowerInfo = struct { +pub const PowerInformation = struct { /// The SMM command port and the value that switches the platform into ACPI mode. smi_cmd: u16 = 0, acpi_enable: u8 = 0, acpi_disable: u8 = 0, /// PM1 control registers — writing SLP_TYP|SLP_EN here enters a sleep state. - pm1a_cnt: RegAccess = .{}, - pm1b_cnt: RegAccess = .{}, + pm1a_cnt: RegisterAccess = .{}, + pm1b_cnt: RegisterAccess = .{}, /// The FADT reset register and the value to write to it. - reset: RegAccess = .{}, + reset: RegisterAccess = .{}, reset_value: u8 = 0, reset_supported: bool = false, /// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`) packages. @@ -56,7 +56,7 @@ pub const PowerInfo = struct { }; /// Filled in by `discover`; the power service reads it to reboot/shutdown. -pub var power_info: PowerInfo = .{}; +pub var power_information: PowerInformation = .{}; /// A legacy ISA IRQ remapped to a different global system interrupt (GSI), from a /// MADT Interrupt Source Override. `flags` are the MPS INTI polarity/trigger bits. @@ -66,11 +66,11 @@ pub const IsoEntry = struct { flags: u16, }; -/// Firmware facts the arch layer needs to avoid legacy assumptions (so danos boots +/// Firmware facts the architecture layer needs to avoid legacy assumptions (so danos boots /// on legacy-free UEFI Class 3 machines). MMIO device *addresses* (HPET, IOAPIC) /// come from the device tree instead; this holds the scalar facts that have no /// natural device node. -pub const PlatformInfo = struct { +pub const PlatformInformation = struct { /// Whether the legacy 8259 PIC is present (MADT flags bit 0, PCAT_COMPAT). When /// false, the PIC must not be programmed (it may not exist). pic_present: bool = false, @@ -78,11 +78,11 @@ pub const PlatformInfo = struct { lapic_base: u64 = 0xFEE00000, /// The ACPI power-management timer — a fixed 3.579545 MHz counter usable as a /// calibration reference when no HPET is present. - pm_timer: RegAccess = .{}, - /// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VAL_EXT). + pm_timer: RegisterAccess = .{}, + /// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VALUE_EXT). pm_timer_32bit: bool = false, /// The console UART the firmware points at (SPCR), if any — MMIO or I/O port. - spcr_uart: ?RegAccess = null, + spcr_uart: ?RegisterAccess = null, /// SPCR interface type (0/1 = 16550/16450, …). spcr_kind: u8 = 0, /// ISA-IRQ-to-GSI remappings from the MADT (for future IOAPIC routing). @@ -90,8 +90,8 @@ pub const PlatformInfo = struct { override_count: usize = 0, }; -/// Filled in by `discover`; the arch layer reads it during bring-up. -pub var platform_info: PlatformInfo = .{}; +/// Filled in by `discover`; the architecture layer reads it during bring-up. +pub var platform_information: PlatformInformation = .{}; /// One usable logical processor, from a MADT type-0 (Local APIC) record. The /// `apic_id` is the Local APIC ID that SMP bring-up targets to wake this core @@ -108,19 +108,19 @@ pub const Cpu = struct { /// The set of usable logical processors the MADT listed — the hardware's degree of /// parallelism. Includes the bootstrap processor danos already runs on; the rest /// are the application processors SMP bring-up would start (see docs/smp.md). -pub const CpuInfo = struct { +pub const CpuInformation = struct { /// A static pool sized well above any danos target (a desktop, two 4-core Pis). /// If the MADT ever lists more, the surplus is dropped and counted in `dropped` /// so the truncation is never silent. - cpus: [max_cpus]Cpu = undefined, + cpus: [maximum_cpus]Cpu = undefined, count: usize = 0, dropped: usize = 0, }; -const max_cpus = config.max_cpus; +const maximum_cpus = parameters.maximum_cpus; /// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs. -pub var cpu_info: CpuInfo = .{}; +pub var cpu_information: CpuInformation = .{}; /// Integrity/diagnostics for the AML parse. `consumed == total` means the parser /// walked every byte of the DSDT/SSDTs without desyncing. @@ -136,20 +136,20 @@ pub var aml_stats: AmlStats = .{}; pub var namespace: ?aml.Namespace = null; /// Physical address of the DSDT the FADT points at, or 0. -pub var dsdt_phys: u64 = 0; +pub var dsdt_physical: u64 = 0; // AML blocks (DSDT + any SSDTs) collected during the table walk, as physical // address + length of each table's post-header bytecode. Scanned after the walk // for the sleep-state (`_Sx`) packages. -var aml_block_phys: [32]u64 = undefined; +var aml_block_physical: [32]u64 = undefined; var aml_block_len: [32]usize = undefined; var aml_block_count: usize = 0; -fn addAmlBlock(sdt_phys: u64) void { - if (aml_block_count >= aml_block_phys.len or sdt_phys == 0) return; - const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys)); +fn addAmlBlock(sdt_physical: u64) void { + if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return; + const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(sdt_physical)); if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return; - aml_block_phys[aml_block_count] = sdt_phys + @sizeOf(SystemDescriptorTableHeader); + aml_block_physical[aml_block_count] = sdt_physical + @sizeOf(SystemDescriptorTableHeader); aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader); aml_block_count += 1; } @@ -362,47 +362,47 @@ const PciHeader = extern struct { // --- Entry point ------------------------------------------------------------ -/// Discover hardware from the ACPI tables rooted at `rsdp_phys` and populate -/// `dt`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the -/// FADT and the AML sleep-state (`_Sx`) packages into `power_info` for the power service. -pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void { - if (rsdp_phys == 0) return error.NoRsdp; +/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate +/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the +/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service. +pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void { + if (rsdp_physical == 0) return error.NoRsdp; // Start clean so a re-run doesn't accumulate stale state. - power_info = .{}; - platform_info = .{}; + power_information = .{}; + platform_information = .{}; aml_stats = .{}; namespace = null; - dsdt_phys = 0; + dsdt_physical = 0; aml_block_count = 0; - const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physToVirt(rsdp_phys)); + const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physicalToVirtual(rsdp_physical)); if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature; // Revision 0 checksums only the first 20 bytes (the v1.0 RSDP). - if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), 20)) return error.BadRsdpChecksum; + if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum; if (rsdp.revision >= 2) { - const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physToVirt(rsdp_phys)); - if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), xsdp.length)) return error.BadXsdpChecksum; - try walkRoot(u64, xsdp.extended_system_descriptor_table_address, dt, hal); + const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physicalToVirtual(rsdp_physical)); + if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum; + try walkRoot(u64, xsdp.extended_system_descriptor_table_address, device_tree, hal); } else { - try walkRoot(u32, rsdp.root_system_description_table_address, dt, hal); + try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal); } // Now that the DSDT and any SSDTs are collected, build the AML namespace and // read the sleep types from it. - var blocks: [aml_block_phys.len][]const u8 = undefined; + var blocks: [aml_block_physical.len][]const u8 = undefined; for (0..aml_block_count) |i| { - blocks[i] = @as([*]const u8, @ptrFromInt(danos.physToVirt(aml_block_phys[i])))[0..aml_block_len[i]]; + blocks[i] = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]]; } const active = blocks[0..aml_block_count]; - if (aml.parse(dt.allocator, active)) |pr| { + if (aml.parse(device_tree.allocator, active)) |pr| { namespace = pr.namespace; aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total }; - power_info.s5 = aml.sleepState(&namespace.?, 5); - power_info.s3 = aml.sleepState(&namespace.?, 3); + power_information.s5 = aml.sleepState(&namespace.?, 5); + power_information.s3 = aml.sleepState(&namespace.?, 3); // Fold the namespace's Device objects into the generic tree. - wireAcpiDevices(dt, &namespace.?, hal) catch {}; + wireAcpiDevices(device_tree, &namespace.?, hal) catch {}; } else |_| { // AML parse failed (e.g. out of memory); power stays best-effort with // whatever the FADT alone provided. @@ -411,51 +411,51 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void { /// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch /// each SDT it points at. A bad individual table is skipped, not fatal. -fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !void { - const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(root_phys)); - if (!checksumOk(@ptrFromInt(danos.physToVirt(root_phys)), header.length)) return error.BadRootChecksum; +fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void { + const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(root_physical)); + if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum; const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry); - const base: [*]const u8 = @ptrFromInt(danos.physToVirt(root_phys)); + const base: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(root_physical)); const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader)); for (entries[0..count]) |ent| { - const sdt_phys: u64 = ent; // u32 entries widen; u64 pass through - handleTable(dt, hal, sdt_phys) catch continue; + const sdt_physical: u64 = ent; // u32 entries widen; u64 pass through + handleTable(device_tree, hal, sdt_physical) catch continue; } } /// Dispatch a single SDT on its signature. -fn handleTable(dt: *DeviceTree, hal: Hal, sdt_phys: u64) !void { - const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys)); +fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void { + const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(sdt_physical)); const sig = header.signature; if (std.mem.eql(u8, &sig, &APIC)) { - try parseMadt(dt, header); + try parseMadt(device_tree, header); } else if (std.mem.eql(u8, &sig, &MCFG)) { - try parseMcfg(dt, hal, header); + try parseMcfg(device_tree, hal, header); } else if (std.mem.eql(u8, &sig, &HPET)) { - try parseHpet(dt, header); + try parseHpet(device_tree, hal, header); } else if (std.mem.eql(u8, &sig, &FACP)) { parseFadt(header); } else if (std.mem.eql(u8, &sig, &SPCR)) { parseSpcr(header); } else if (std.mem.eql(u8, &sig, &SSDT)) { // Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan. - addAmlBlock(sdt_phys); + addAmlBlock(sdt_physical); } // Any other signature is recognised but left opaque for now. } /// MADT -> one processor node per Local APIC, one interrupt_controller per I/O APIC. -fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void { +fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void { const madt: *const Madt = @ptrCast(header); const total: usize = header.length; const base: [*]const u8 = @ptrCast(header); var ioapic_index: usize = 0; // MADT header: local APIC base + flags (bit 0 = 8259 PIC present). - platform_info.lapic_base = madt.local_apic_address; - platform_info.pic_present = madt.flags & 1 != 0; + platform_information.lapic_base = madt.local_apic_address; + platform_information.pic_present = madt.flags & 1 != 0; var off: usize = @sizeOf(Madt); while (off + @sizeOf(MadtRecordHeader) <= total) { @@ -468,18 +468,18 @@ fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void if (la.flags & 1 != 0) { var nb: [24]u8 = undefined; const nm = std.fmt.bufPrint(&nb, "cpu{d}", .{la.processor_id}) catch "cpu"; - _ = try dt.addChild(dt.root, .processor, nm); + _ = try device_tree.addChild(device_tree.root, .processor, nm); // Also record it as a schedulable core (with the APIC ID an AP // wake needs, which the device node name doesn't preserve). - if (cpu_info.count < cpu_info.cpus.len) { - cpu_info.cpus[cpu_info.count] = .{ + if (cpu_information.count < cpu_information.cpus.len) { + cpu_information.cpus[cpu_information.count] = .{ .processor_id = la.processor_id, .apic_id = la.apic_id, .online_capable = la.flags & 2 != 0, }; - cpu_info.count += 1; + cpu_information.count += 1; } else { - cpu_info.dropped += 1; + cpu_information.dropped += 1; } } }, @@ -488,25 +488,25 @@ fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void var nb: [24]u8 = undefined; const nm = std.fmt.bufPrint(&nb, "ioapic{d}", .{ioapic_index}) catch "ioapic"; ioapic_index += 1; - const d = try dt.addChild(dt.root, .interrupt_controller, nm); + const d = try device_tree.addChild(device_tree.root, .interrupt_controller, nm); _ = d.addResource(.memory, io.address, 0x20); // The GSI range this I/O APIC handles, starting at gsi_base. _ = d.addResource(.irq, io.gsi_base, 0); }, 2 => { const iso: *const MadtIso = @ptrCast(base + off); - if (platform_info.override_count < platform_info.overrides.len) { - platform_info.overrides[platform_info.override_count] = .{ + if (platform_information.override_count < platform_information.overrides.len) { + platform_information.overrides[platform_information.override_count] = .{ .source = iso.source, .gsi = iso.gsi, .flags = iso.flags, }; - platform_info.override_count += 1; + platform_information.override_count += 1; } }, 5 => { const ovr: *const MadtLapicOverride = @ptrCast(base + off); - platform_info.lapic_base = ovr.address; + platform_information.lapic_base = ovr.address; }, else => {}, } @@ -515,7 +515,7 @@ fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void } /// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath. -fn parseMcfg(dt: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void { +fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void { const total: usize = header.length; const base: [*]const u8 = @ptrCast(header); @@ -526,12 +526,12 @@ fn parseMcfg(dt: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHead var nb: [24]u8 = undefined; const nm = std.fmt.bufPrint(&nb, "pci{d}", .{alloc.segment_group}) catch "pci"; - const bridge = try dt.addChild(dt.root, .pci_host_bridge, nm); - // ECAM window: 1 MiB of config space per bus. + const bridge = try device_tree.addChild(device_tree.root, .pci_host_bridge, nm); + // ECAM window: 1 MiB of configuration space per bus. _ = bridge.addResource(.memory, alloc.base_address, bus_count << 20); _ = bridge.addResource(.bus_range, alloc.start_bus, bus_count); - try enumeratePci(dt, bridge, hal, alloc.*); + try enumeratePci(device_tree, bridge, hal, alloc.*); } } @@ -539,38 +539,38 @@ fn parseMcfg(dt: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHead /// bridge recursion yet: on the ECAM path the host bridge decodes every bus in /// the window, so scanning the declared range finds everything QEMU exposes. fn enumeratePci( - dt: *DeviceTree, - bridge: *device.Device, + device_tree: *DeviceTree, + bridge: *device_model.Device, hal: Hal, alloc: McfgAllocation, ) !void { var bus: u16 = alloc.start_bus; while (bus <= alloc.end_bus) : (bus += 1) { - var dev: u8 = 0; - while (dev < 32) : (dev += 1) { - const h0: *align(1) const PciHeader = @ptrCast(pciConfigPtr(alloc, hal, @intCast(bus), dev, 0)); + var device: u8 = 0; + while (device < 32) : (device += 1) { + const h0: *align(1) const PciHeader = @ptrCast(pciConfigurationPtr(alloc, hal, @intCast(bus), device, 0)); if (h0.vendor_id == 0xFFFF) continue; // no function 0 => slot empty const funcs: u8 = if (h0.header_type & 0x80 != 0) 8 else 1; - var func: u8 = 0; - while (func < funcs) : (func += 1) { - const cfg = pciConfigPtr(alloc, hal, @intCast(bus), dev, func); - const h: *align(1) const PciHeader = @ptrCast(cfg); + var function: u8 = 0; + while (function < funcs) : (function += 1) { + const configuration = pciConfigurationPtr(alloc, hal, @intCast(bus), device, function); + const h: *align(1) const PciHeader = @ptrCast(configuration); if (h.vendor_id == 0xFFFF) continue; var nb: [24]u8 = undefined; const nm = std.fmt.bufPrint(&nb, "{s}:{x:0>2}:{x:0>2}.{d}", .{ - bridge.name(), bus, dev, func, + bridge.name(), bus, device, function, }) catch "pcidev"; - const node = try dt.addChild(bridge, .pci_device, nm); + const node = try device_tree.addChild(bridge, .pci_device, nm); node.ids.pci_vendor = h.vendor_id; node.ids.pci_device = h.device_id; node.ids.pci_class = (@as(u24, h.class_code) << 16) | (@as(u24, h.subclass) << 8) | h.prog_if; - node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, dev) << 3) | func; + node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, device) << 3) | function; // BARs only exist in header type 0 (normal devices), not bridges. - if (h.header_type & 0x7F == 0) addBars(node, cfg); + if (h.header_type & 0x7F == 0) addBars(node, configuration); } } } @@ -579,58 +579,96 @@ fn enumeratePci( /// Record and size the memory/IO windows named by a device's Base Address /// Registers. Sizing is the standard probe: disable decode, write all-ones, read /// back the writable (address) bits, restore. `size = ~mask + 1`. -fn addBars(node: *device.Device, cfg: [*]align(1) u8) void { +fn addBars(node: *device_model.Device, configuration: [*]align(1) u8) void { // Stop the device decoding its BARs while we transiently write all-ones. - const command = rd(u16, cfg, 0x04); - wr(u16, cfg, 0x04, command & ~@as(u16, 0b11)); + const command = rd(u16, configuration, 0x04); + wr(u16, configuration, 0x04, command & ~@as(u16, 0b11)); var i: usize = 0; while (i < 6) : (i += 1) { const off = 0x10 + i * 4; - const orig = rd(u32, cfg, off); + const orig = rd(u32, configuration, off); if (orig == 0) continue; if (orig & 1 != 0) { // I/O-space BAR (16-bit address space on x86). - wr(u32, cfg, off, 0xFFFF_FFFF); - const readback = rd(u32, cfg, off); - wr(u32, cfg, off, orig); + wr(u32, configuration, off, 0xFFFF_FFFF); + const readback = rd(u32, configuration, off); + wr(u32, configuration, off, orig); const mask = readback & 0xFFFF_FFFC; const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF; _ = node.addResource(.io_port, orig & 0xFFFF_FFFC, size); } else if ((orig >> 1) & 0x3 == 2) { - // 64-bit memory BAR: this BAR pair spans two config slots. - const orig_hi = rd(u32, cfg, off + 4); - wr(u32, cfg, off, 0xFFFF_FFFF); - wr(u32, cfg, off + 4, 0xFFFF_FFFF); - const lo = rd(u32, cfg, off); - const hi = rd(u32, cfg, off + 4); - wr(u32, cfg, off, orig); - wr(u32, cfg, off + 4, orig_hi); + // 64-bit memory BAR: this BAR pair spans two configuration slots. + const orig_hi = rd(u32, configuration, off + 4); + wr(u32, configuration, off, 0xFFFF_FFFF); + wr(u32, configuration, off + 4, 0xFFFF_FFFF); + const lo = rd(u32, configuration, off); + const hi = rd(u32, configuration, off + 4); + wr(u32, configuration, off, orig); + wr(u32, configuration, off + 4, orig_hi); const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0); const size: u64 = if (readback == 0) 0 else ~readback +% 1; - const addr = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0); - _ = node.addResource(.memory, addr, size); + const address = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0); + _ = node.addResource(.memory, address, size); i += 1; // consumed the high half } else { // 32-bit memory BAR. - wr(u32, cfg, off, 0xFFFF_FFFF); - const readback = rd(u32, cfg, off); - wr(u32, cfg, off, orig); + wr(u32, configuration, off, 0xFFFF_FFFF); + const readback = rd(u32, configuration, off); + wr(u32, configuration, off, orig); const mask = readback & 0xFFFF_FFF0; const size: u32 = if (mask == 0) 0 else ~mask +% 1; _ = node.addResource(.memory, orig & 0xFFFF_FFF0, size); } } - wr(u16, cfg, 0x04, command); // restore decode + wr(u16, configuration, 0x04, command); // restore decode } -/// HPET -> a timer node with its register block as an MMIO resource. -fn parseHpet(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void { +/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI +/// its comparators can raise. +/// +/// Unlike a PCI device or an ACPI `_CRS` node, the HPET table carries **no interrupt +/// number**: which I/O APIC inputs a comparator may drive is advertised at runtime, +/// as a bitmask in `Tn_INT_ROUTE_CAP` (bits 63:32 of the Timer 0 configuration register). +/// So discovery maps the register block, reads the mask, and records one concrete +/// `irq` resource — the GSI a driver is entitled to bind. The driver commits to it +/// by writing `Tn_INT_ROUTE_CNF`; the kernel checks the binding against this +/// resource (see process.ownedGsi), which is what keeps `irq_bind` a capability +/// rather than a request for an arbitrary interrupt line. +fn parseHpet(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void { const hpet: *const Hpet = @ptrCast(header); - const d = try dt.addChild(dt.root, .timer, "hpet"); + const d = try device_tree.addChild(device_tree.root, .timer, "hpet"); + + // The GAS tag must say System Memory (0) before we treat `address` as a physical + // address. The HPET spec mandates it, but firmware is not a thing to trust: a + // System I/O (1) tag here would have us map an arbitrary page and read a bogus + // route-capability mask out of it. + if (hpet.address_space_id != gas_system_memory) return; + _ = d.addResource(.memory, hpet.address, 0x400); + + const regs = hal.mapMmio(hpet.address, 0x400, true); + const t0_configuration: *const volatile u64 = @ptrFromInt(regs + 0x100); + const route_cap: u32 = @truncate(t0_configuration.* >> 32); + if (hpetGsi(route_cap)) |gsi| _ = d.addResource(.irq, gsi, 1); +} + +/// ACPI Generic Address Structure address-space ids we care about. +const gas_system_memory: u8 = 0; + +/// Pick a GSI for the HPET out of its route-capability mask. Prefer an input at or +/// above 16: the low ones overlap the legacy ISA lines (2 = cascaded PIT, 8 = RTC), +/// which the MADT may separately override, whereas 16+ are the free upper inputs on +/// every I/O APIC we care about. Falls back to the lowest bit set if there are none. +fn hpetGsi(route_cap: u32) ?u32 { + if (route_cap == 0) return null; + var gsi: u32 = 16; + while (gsi < 32) : (gsi += 1) { + if (route_cap & (@as(u32, 1) << @intCast(gsi)) != 0) return gsi; + } + return @ctz(route_cap); } // FADT field offsets (bytes from the table start). The FADT grew across ACPI @@ -645,45 +683,45 @@ const fadt_pm1b_cnt_blk = 68; // u32 (I/O port) const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter const fadt_pm1_cnt_len = 89; // u8 (bytes) const fadt_flags = 112; // u32 -const fadt_reset_reg = 116; // GAS (12 bytes) +const fadt_reset_register = 116; // GAS (12 bytes) const fadt_reset_value = 128; // u8 const fadt_x_dsdt = 140; // u64 const fadt_x_pm1a_cnt_blk = 172; // GAS const fadt_x_pm1b_cnt_blk = 184; // GAS const fadt_x_pm_tmr_blk = 208; // GAS -const flag_reset_reg_supported = 1 << 10; -const flag_tmr_val_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit) +const flag_reset_register_supported = 1 << 10; +const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit) -/// FADT -> the power register map (into `power_info`) and the DSDT address, which +/// FADT -> the power register map (into `power_information`) and the DSDT address, which /// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here. fn parseFadt(header: *const SystemDescriptorTableHeader) void { const base: [*]align(1) const u8 = @ptrCast(header); const len: usize = header.length; - const pi = &power_info; + const pi = &power_information; pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0); pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0; pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0; const cnt_width = fadt(u8, base, len, fadt_pm1_cnt_len) orelse 2; - pi.pm1a_cnt = readCntReg(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width); - pi.pm1b_cnt = readCntReg(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width); + pi.pm1a_cnt = readCntRegister(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width); + pi.pm1b_cnt = readCntRegister(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width); const flags = fadt(u32, base, len, fadt_flags) orelse 0; - pi.reset_supported = flags & flag_reset_reg_supported != 0; - pi.reset = readGas(base, len, fadt_reset_reg) orelse .{}; + pi.reset_supported = flags & flag_reset_register_supported != 0; + pi.reset = readGas(base, len, fadt_reset_register) orelse .{}; pi.reset_value = fadt(u8, base, len, fadt_reset_value) orelse 0; // The PM timer — a fixed-rate counter used as a calibration reference when no // HPET is present. Prefer the 64-bit-capable X_ GAS, fall back to the port. - platform_info.pm_timer = readCntReg(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4); - platform_info.pm_timer_32bit = flags & flag_tmr_val_ext != 0; + platform_information.pm_timer = readCntRegister(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4); + platform_information.pm_timer_32bit = flags & flag_tmr_value_ext != 0; var dsdt: u64 = fadt(u32, base, len, fadt_dsdt) orelse 0; if (fadt(u64, base, len, fadt_x_dsdt)) |x| { if (x != 0) dsdt = x; } - dsdt_phys = dsdt; + dsdt_physical = dsdt; addAmlBlock(dsdt); } @@ -698,15 +736,15 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void { const len: usize = header.length; const gas = readGas(base, len, spcr_base_address) orelse return; if (gas.address == 0) return; - platform_info.spcr_uart = gas; - platform_info.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0; + platform_information.spcr_uart = gas; + platform_information.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0; } // --- AML namespace -> generic device tree ----------------------------------- /// The PCI bus context while descending the ACPI namespace: the generic host /// bridge whose children ACPI address (`_ADR`) devices resolve against, and the bus number. -const PciCtx = struct { bridge: *device.Device, bus: u8 }; +const PciContext = struct { bridge: *device_model.Device, bus: u8 }; /// Mirror the ACPI namespace's Device objects into the generic tree, *merging* /// them with the PCI-enumerated nodes: a PCI root bridge (`PNP0A03`/`PNP0A08`) @@ -714,73 +752,73 @@ const PciCtx = struct { bridge: *device.Device, bus: u8 }; /// the matching PCI function (annotating it with the ACPI hardware ID (`_HID`) and nesting the /// ACPI-only children — keyboard, RTC, … — beneath it). Namespace devices with no /// PCI match land under a synthetic `acpi` node. -fn wireAcpiDevices(dt: *DeviceTree, nsp: *aml.Namespace, hal: Hal) !void { - var arena = std.heap.ArenaAllocator.init(dt.allocator); +fn wireAcpiDevices(device_tree: *DeviceTree, aml_namespace: *aml.Namespace, hal: Hal) !void { + var arena = std.heap.ArenaAllocator.init(device_tree.allocator); defer arena.deinit(); - var ev = aml.Interp.init(nsp, .{ + var interpreter = aml.Interpreter.init(aml_namespace, .{ .mapMmio = hal.mapMmio, .pioRead = hal.pioRead, .pioWrite = hal.pioWrite, }, arena.allocator()); - const acpi_root = try dt.addChild(dt.root, .unknown, "acpi"); - try mirrorDevices(dt, nsp.root, acpi_root, null, &ev); + const acpi_root = try device_tree.addChild(device_tree.root, .unknown, "acpi"); + try mirrorDevices(device_tree, aml_namespace.root, acpi_root, null, &interpreter); } -fn mirrorDevices(dt: *DeviceTree, node: *aml.Node, parent_dev: *device.Device, ctx: ?PciCtx, ev: *aml.Interp) (error{OutOfMemory})!void { +fn mirrorDevices(device_tree: *DeviceTree, node: *aml.Node, parent_device: *device_model.Device, context: ?PciContext, interpreter: *aml.Interpreter) (error{OutOfMemory})!void { var child = node.first_child; while (child) |c| : (child = c.next_sibling) { if (c.kind != .device) { // A scope — the System Bus (\_SB), General Purpose Events (\_GPE), … — // descend without adding a node. - try mirrorDevices(dt, c, parent_dev, ctx, ev); + try mirrorDevices(device_tree, c, parent_device, context, interpreter); continue; } // Skip devices the firmware reports as not present (via a device-status (`_STA`) method), // along with their whole subtree — per the ACPI rules. - if (!devicePresent(ev, c)) continue; + if (!devicePresent(interpreter, c)) continue; - var gdev: *device.Device = undefined; - var child_ctx = ctx; + var mirrored_device: *device_model.Device = undefined; + var child_context = context; if (isPciRootNode(c)) { // The PCI root bridge folds onto the generic host bridge. - gdev = matchHostBridge(dt) orelse - try dt.addChild(parent_dev, .acpi_device, &c.seg); - child_ctx = .{ .bridge = gdev, .bus = 0 }; + mirrored_device = matchHostBridge(device_tree) orelse + try device_tree.addChild(parent_device, .acpi_device, &c.segment); + child_context = .{ .bridge = mirrored_device, .bus = 0 }; } else { // An addressed device folds onto its matching PCI function; anything // else becomes a fresh node under the current parent. - gdev = pick: { - if (ctx) |pc| { + mirrored_device = pick: { + if (context) |pc| { if (readAdr(c)) |adr| { if (findPciNode(pc.bridge, pc.bus, adr)) |pnode| break :pick pnode; } } - break :pick try dt.addChild(parent_dev, .acpi_device, &c.seg); + break :pick try device_tree.addChild(parent_device, .acpi_device, &c.segment); }; } - applyHid(gdev, c, ev); - applyCrs(gdev, c, ev); - try mirrorDevices(dt, c, gdev, child_ctx, ev); + applyHid(mirrored_device, c, interpreter); + applyCrs(mirrored_device, c, interpreter); + try mirrorDevices(device_tree, c, mirrored_device, child_context, interpreter); } } /// Evaluate a device's status (`_STA`) to decide if it is present. An absent status /// (`_STA`) means present by default; an evaluation failure is treated as present too (we'd /// rather over-report than hide a device we couldn't introspect). -fn devicePresent(ev: *aml.Interp, node: *aml.Node) bool { +fn devicePresent(interpreter: *aml.Interpreter, node: *aml.Node) bool { const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true; - const obj = ev.evaluate(sta, &.{}) catch return true; - const status = obj.asInt() catch return true; + const obj = interpreter.evaluate(sta, &.{}) catch return true; + const status = obj.asInteger() catch return true; return (status & 0x01) != 0; // bit 0 = present } /// The first PCI host bridge in the generic tree (segment 0). -fn matchHostBridge(dt: *DeviceTree) ?*device.Device { - var c = dt.root.first_child; +fn matchHostBridge(device_tree: *DeviceTree) ?*device_model.Device { + var c = device_tree.root.first_child; while (c) |ch| : (c = ch.next_sibling) { if (ch.class == .pci_host_bridge) return ch; } @@ -788,12 +826,12 @@ fn matchHostBridge(dt: *DeviceTree) ?*device.Device { } /// The PCI function node under `bridge` at the address the device's address object -/// (`_ADR`) names (dev/func on +/// (`_ADR`) names (device/function on /// `bus`), or null. -fn findPciNode(bridge: *device.Device, bus: u8, adr: u32) ?*device.Device { - const dev: u16 = @truncate((adr >> 16) & 0x1F); - const func: u16 = @truncate(adr & 0x7); - const target: u16 = (@as(u16, bus) << 8) | (dev << 3) | func; +fn findPciNode(bridge: *device_model.Device, bus: u8, adr: u32) ?*device_model.Device { + const device: u16 = @truncate((adr >> 16) & 0x1F); + const function: u16 = @truncate(adr & 0x7); + const target: u16 = (@as(u16, bus) << 8) | (device << 3) | function; var c = bridge.first_child; while (c) |ch| : (c = ch.next_sibling) { if (ch.ids.pci_bdf) |bdf| { @@ -832,13 +870,13 @@ fn isPciRootNode(node: *aml.Node) bool { /// Read a device's hardware ID (`_HID`) into the generic device: an integer decodes as an EISA /// id ("PNP0A03"), a string is taken verbatim. Handles both the common static /// Name form and a Method form (evaluated). -fn applyHid(dev: *device.Device, node: *aml.Node, ev: *aml.Interp) void { +fn applyHid(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void { const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return; if (hid.kind == .method) { - const obj = ev.evaluate(hid, &.{}) catch return; + const obj = interpreter.evaluate(hid, &.{}) catch return; switch (obj) { - .integer => |n| setEisaHid(dev, @truncate(n)), - .string => |s| dev.setHid(s), + .integer => |n| setEisaHid(device, @truncate(n)), + .string => |s| device.setHid(s), else => {}, } return; @@ -849,34 +887,34 @@ fn applyHid(dev: *device.Device, node: *aml.Node, ev: *aml.Interp) void { 0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => { var p: usize = 0; const n = readIntObj(v, &p) orelse return; - setEisaHid(dev, @truncate(n)); + setEisaHid(device, @truncate(n)); }, - 0x0D => dev.setHid(cstr(v[1..])), // StringPrefix + 0x0D => device.setHid(cstr(v[1..])), // StringPrefix else => {}, } } -fn setEisaHid(dev: *device.Device, id: u32) void { - dev.ids.acpi_hid = id; - var buf: [8]u8 = undefined; - dev.setHid(eisaIdToStr(id, &buf)); +fn setEisaHid(device: *device_model.Device, id: u32) void { + device.ids.acpi_hid = id; + var buffer: [8]u8 = undefined; + device.setHid(eisaIdToStr(id, &buffer)); } /// Parse a device's current resource settings (`_CRS`). The evaluator handles both the static /// `Buffer` form (a `Name`) and the method form uniformly, yielding the /// ResourceTemplate bytes we then decode. -fn applyCrs(dev: *device.Device, node: *aml.Node, ev: *aml.Interp) void { +fn applyCrs(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void { const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return; - const obj = ev.evaluate(crs, &.{}) catch return; - const buf = switch (obj) { + const obj = interpreter.evaluate(crs, &.{}) catch return; + const buffer = switch (obj) { .buffer => |b| b, else => return, }; - parseResourceTemplate(dev, buf); + parseResourceTemplate(device, buffer); } /// Walk a ResourceTemplate byte list, adding recognised descriptors as resources. -fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void { +fn parseResourceTemplate(device: *device_model.Device, bytes: []const u8) void { var i: usize = 0; while (i < bytes.len) { const tag = bytes[i]; @@ -890,14 +928,14 @@ fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void { const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8); var b: usize = 0; while (b < 16) : (b += 1) { - if (mask & (@as(u16, 1) << @intCast(b)) != 0) _ = dev.addResource(.irq, b, 1); + if (mask & (@as(u16, 1) << @intCast(b)) != 0) _ = device.addResource(.irq, b, 1); } }, - 0x08 => if (len >= 7) { // IO port: min at +1, length at +6 - _ = dev.addResource(.io_port, rd16(bytes, body + 1), bytes[body + 6]); + 0x08 => if (len >= 7) { // IO port: minimum at +1, length at +6 + _ = device.addResource(.io_port, rd16(bytes, body + 1), bytes[body + 6]); }, 0x09 => if (len >= 3) { // Fixed IO: base at +0, length at +2 - _ = dev.addResource(.io_port, rd16(bytes, body), bytes[body + 2]); + _ = device.addResource(.io_port, rd16(bytes, body), bytes[body + 2]); }, 0x0F => break, // EndTag else => {}, @@ -910,20 +948,20 @@ fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void { const body = i + 3; if (body + len > bytes.len) break; switch (tag) { - 0x85 => if (len >= 17) { // Memory32: min at +1, length at +13 - _ = dev.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 13)); + 0x85 => if (len >= 17) { // Memory32: minimum at +1, length at +13 + _ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 13)); }, 0x86 => if (len >= 9) { // Memory32Fixed: base at +1, length at +5 - _ = dev.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 5)); + _ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 5)); }, 0x89 => if (len >= 2) { // Extended IRQ: count at +1, then count u32s const count = bytes[body + 1]; var k: usize = 0; while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) { - _ = dev.addResource(.irq, rd32(bytes, body + 2 + k * 4), 1); + _ = device.addResource(.irq, rd32(bytes, body + 2 + k * 4), 1); } }, - 0x87, 0x88, 0x8A => parseAddressSpace(dev, tag, bytes[body .. body + len]), + 0x87, 0x88, 0x8A => parseAddressSpace(device, tag, bytes[body .. body + len]), else => {}, } i = body + len; @@ -932,39 +970,39 @@ fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void { } /// Word/DWord/QWord address-space descriptors: resource type at [0], then -/// granularity/min/max/translation/length, each of width `w`. -fn parseAddressSpace(dev: *device.Device, tag: u8, body: []const u8) void { +/// granularity/minimum/maximum/translation/length, each of width `w`. +fn parseAddressSpace(device: *device_model.Device, tag: u8, body: []const u8) void { const w: usize = switch (tag) { 0x88 => 2, // Word 0x87 => 4, // DWord else => 8, // QWord (0x8A) }; if (body.len < 3 + 5 * w) return; - const min = readN(body, 3 + w, w); + const minimum = readN(body, 3 + w, w); const length = readN(body, 3 + 4 * w, w); - const kind: device.ResourceKind = switch (body[0]) { + const kind: device_model.ResourceKind = switch (body[0]) { 0 => .memory, 1 => .io_port, else => .bus_range, }; - _ = dev.addResource(kind, min, length); + _ = device.addResource(kind, minimum, length); } /// Decode a packed EISA id into its 7-char string (e.g. 0x030AD041 -> "PNP0A03"). -fn eisaIdToStr(id: u32, buf: *[8]u8) []const u8 { +fn eisaIdToStr(id: u32, buffer: *[8]u8) []const u8 { const b0: u16 = @intCast(id & 0xFF); const b1: u16 = @intCast((id >> 8) & 0xFF); const b2: u8 = @truncate(id >> 16); const b3: u8 = @truncate(id >> 24); const mfg = (b0 << 8) | b1; - buf[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F)); - buf[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F)); - buf[2] = '@' + @as(u8, @intCast(mfg & 0x1F)); - buf[3] = hexDigit((b2 >> 4) & 0xF); - buf[4] = hexDigit(b2 & 0xF); - buf[5] = hexDigit((b3 >> 4) & 0xF); - buf[6] = hexDigit(b3 & 0xF); - return buf[0..7]; + buffer[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F)); + buffer[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F)); + buffer[2] = '@' + @as(u8, @intCast(mfg & 0x1F)); + buffer[3] = hexDigit((b2 >> 4) & 0xF); + buffer[4] = hexDigit(b2 & 0xF); + buffer[5] = hexDigit((b3 >> 4) & 0xF); + buffer[6] = hexDigit(b3 & 0xF); + return buffer[0..7]; } fn hexDigit(n: u8) u8 { @@ -976,13 +1014,13 @@ fn seg4(comptime s: *const [4:0]u8) [4]u8 { } fn cstr(bytes: []const u8) []const u8 { - const idx = std.mem.indexOfScalar(u8, bytes, 0) orelse bytes.len; - return bytes[0..idx]; + const index = std.mem.indexOfScalar(u8, bytes, 0) orelse bytes.len; + return bytes[0..index]; } const PkgLen = struct { value: usize, size: usize }; -fn pkgLen(bytes: []const u8, p: usize) ?PkgLen { +fn packageLength(bytes: []const u8, p: usize) ?PkgLen { if (p >= bytes.len) return null; const lead = bytes[p]; const follow: usize = lead >> 6; @@ -1049,9 +1087,9 @@ fn fadt(comptime T: type, base: [*]align(1) const u8, len: usize, off: usize) ?T return rd(T, base, off); } -/// Decode a Generic Address Structure at `off` into a `RegAccess`. GAS layout: +/// Decode a Generic Address Structure at `off` into a `RegisterAccess`. GAS layout: /// address_space(u8), bit_width(u8), bit_offset(u8), access_size(u8), address(u64). -fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegAccess { +fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegisterAccess { if (off + 12 > len) return null; const address_space = rd(u8, base, off); const bit_width = rd(u8, base, off + 1); @@ -1065,7 +1103,7 @@ fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegAccess { /// A PM1 control register: prefer the 64-bit-capable X_ GAS form; fall back to the /// legacy 32-bit I/O-port field. Width comes from PM1_CNT_LEN either way. -fn readCntReg(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegAccess { +fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegisterAccess { if (readGas(base, len, xoff)) |g| { if (g.address != 0) return .{ .mmio = g.mmio, .address = g.address, .width = width }; } @@ -1073,16 +1111,16 @@ fn readCntReg(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: u return .{ .mmio = false, .address = port, .width = width }; } -/// The mapped config space of one PCI function (its 4 KiB ECAM page). Mapped +/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped /// writable so BAR sizing can probe it; reads and writes both go through here. -fn pciConfigPtr(alloc: McfgAllocation, hal: Hal, bus: u8, dev: u8, func: u8) [*]align(1) u8 { - const phys = alloc.base_address + +fn pciConfigurationPtr(alloc: McfgAllocation, hal: Hal, bus: u8, device: u8, function: u8) [*]align(1) u8 { + const physical = alloc.base_address + (@as(u64, bus - alloc.start_bus) << 20) + - (@as(u64, dev) << 15) + - (@as(u64, func) << 12); - // Map the config page (writable, for BAR sizing) and use the virtual + (@as(u64, device) << 15) + + (@as(u64, function) << 12); + // Map the configuration page (writable, for BAR sizing) and use the virtual // address the HAL hands back. - return @ptrFromInt(hal.mapMmio(phys, danos.page_size, true)); + return @ptrFromInt(hal.mapMmio(physical, danos.page_size, true)); } /// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer. @@ -1101,30 +1139,30 @@ fn wr(comptime T: type, bytes: [*]align(1) u8, off: usize, value: T) void { // --- tests ------------------------------------------------------------------ test "eisaIdToStr decodes a packed EISA id" { - var buf: [8]u8 = undefined; + var buffer: [8]u8 = undefined; // 0x030AD041 is the well-known encoding of "PNP0A03" (PCI root bridge). - try std.testing.expectEqualStrings("PNP0A03", eisaIdToStr(0x030AD041, &buf)); + try std.testing.expectEqualStrings("PNP0A03", eisaIdToStr(0x030AD041, &buffer)); } test "parseResourceTemplate extracts IO, IRQ, and fixed memory" { - // ResourceTemplate { IO(min 0x60, len 8), IRQ(4), Memory32Fixed(0xFED00000, 0x1000) } - const rt = [_]u8{ + // ResourceTemplate { IO(minimum 0x60, len 8), IRQ(4), Memory32Fixed(0xFED00000, 0x1000) } + const runtime = [_]u8{ 0x47, 0x01, 0x60, 0x00, 0x60, 0x00, 0x01, 0x08, // small IO descriptor 0x22, 0x10, 0x00, // small IRQ descriptor (mask bit 4 -> IRQ 4) 0x86, 0x09, 0x00, 0x01, 0x00, 0x00, 0xD0, 0xFE, 0x00, 0x10, 0x00, 0x00, // Memory32Fixed 0x79, 0x00, // EndTag }; - var dev = device.Device{}; - parseResourceTemplate(&dev, &rt); + var device = device_model.Device{}; + parseResourceTemplate(&device, &runtime); - try std.testing.expectEqual(@as(u8, 3), dev.resource_count); - const rs = dev.resources[0..dev.resource_count]; - try std.testing.expectEqual(device.ResourceKind.io_port, rs[0].kind); + try std.testing.expectEqual(@as(u8, 3), device.resource_count); + const rs = device.resources[0..device.resource_count]; + try std.testing.expectEqual(device_model.ResourceKind.io_port, rs[0].kind); try std.testing.expectEqual(@as(u64, 0x60), rs[0].start); try std.testing.expectEqual(@as(u64, 8), rs[0].len); - try std.testing.expectEqual(device.ResourceKind.irq, rs[1].kind); + try std.testing.expectEqual(device_model.ResourceKind.irq, rs[1].kind); try std.testing.expectEqual(@as(u64, 4), rs[1].start); - try std.testing.expectEqual(device.ResourceKind.memory, rs[2].kind); + try std.testing.expectEqual(device_model.ResourceKind.memory, rs[2].kind); try std.testing.expectEqual(@as(u64, 0xFED00000), rs[2].start); try std.testing.expectEqual(@as(u64, 0x1000), rs[2].len); } diff --git a/src/device/aml/aml.zig b/src/device/aml/aml.zig index 4ec6ab1..79c9464 100644 --- a/src/device/aml/aml.zig +++ b/src/device/aml/aml.zig @@ -9,20 +9,18 @@ //! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace. const std = @import("std"); -const op = @import("opcodes.zig"); +const opcode = @import("opcodes.zig"); const parser = @import("parser.zig"); -const namespace = @import("namespace.zig"); -const interp = @import("interp.zig"); -pub const Namespace = namespace.Namespace; -pub const Node = namespace.Node; -pub const NodeKind = namespace.NodeKind; +pub const Namespace = @import("namespace.zig").Namespace; +pub const Node = @import("namespace.zig").Node; +pub const NodeKind = @import("namespace.zig").NodeKind; /// The AML evaluator: interprets control methods (and reads Names/Fields) far /// enough for device discovery. See `interp.zig`. -pub const Interp = interp.Interp; -pub const Object = interp.Object; -pub const EvalHal = interp.Hal; +pub const Interpreter = @import("interp.zig").Interpreter; +pub const Object = @import("interp.zig").Object; +pub const EvaluateHal = @import("interp.zig").Hal; /// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state. pub const SleepType = struct { @@ -41,22 +39,22 @@ pub const ParseResult = struct { /// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later /// blocks extend the namespace built by earlier ones, exactly as ACPI intends. pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult { - var ns = try Namespace.init(allocator); + var namespace = try Namespace.init(allocator); var consumed: usize = 0; var total: usize = 0; for (blocks) |block| { - var p = parser.Parser.init(block, &ns); + var p = parser.Parser.init(block, &namespace); consumed += p.parseAll(); total += block.len; } - return .{ .namespace = ns, .consumed = consumed, .total = total }; + return .{ .namespace = namespace, .consumed = consumed, .total = total }; } /// Look up the `\_S{state}` sleep package in a parsed namespace and return its /// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent. -pub fn sleepState(ns: *Namespace, state: u8) ?SleepType { - const seg = [4]u8{ '_', 'S', '0' + state, '_' }; - const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null; +pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType { + const segment = [4]u8{ '_', 'S', '0' + state, '_' }; + const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null; if (node.kind != .name) return null; return parseSleepPackage(node.value); } @@ -64,20 +62,20 @@ pub fn sleepState(ns: *Namespace, state: u8) ?SleepType { /// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's /// value. Returns the first two elements as bytes (missing elements default to 0). fn parseSleepPackage(value: []const u8) ?SleepType { - if (value.len == 0 or value[0] != op.package_op) return null; + if (value.len == 0 or value[0] != opcode.package_opcode) return null; var p: usize = 1; - p += pkgLengthSize(value, p) orelse return null; + p += packageLengthSize(value, p) orelse return null; if (p >= value.len) return null; - const num_elements = value[p]; + const number_elements = value[p]; p += 1; - const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0; - const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0; + const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0; + const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0; return .{ .slp_typ_a = a, .slp_typ_b = b }; } /// Bytes a PkgLength field occupies at `p` (we only need to step over it here). -fn pkgLengthSize(bytes: []const u8, p: usize) ?usize { +fn packageLengthSize(bytes: []const u8, p: usize) ?usize { if (p >= bytes.len) return null; const follow: usize = bytes[p] >> 6; if (p + 1 + follow > bytes.len) return null; @@ -87,16 +85,16 @@ fn pkgLengthSize(bytes: []const u8, p: usize) ?usize { /// Read one AML integer data object at `p`, advancing `p`. fn readInteger(bytes: []const u8, p: *usize) ?u64 { if (p.* >= bytes.len) return null; - const opcode = bytes[p.*]; + const opcode_byte = bytes[p.*]; p.* += 1; - return switch (opcode) { - op.zero_op => 0, - op.one_op => 1, - op.ones_op => 0xFF, - op.byte_prefix => readLittle(bytes, p, 1), - op.word_prefix => readLittle(bytes, p, 2), - op.dword_prefix => readLittle(bytes, p, 4), - op.qword_prefix => readLittle(bytes, p, 8), + return switch (opcode_byte) { + opcode.zero_opcode => 0, + opcode.one_opcode => 1, + opcode.ones_opcode => 0xFF, + opcode.byte_prefix => readLittle(bytes, p, 1), + opcode.word_prefix => readLittle(bytes, p, 2), + opcode.dword_prefix => readLittle(bytes, p, 4), + opcode.qword_prefix => readLittle(bytes, p, 8), else => null, }; } @@ -125,19 +123,19 @@ test "parses a nested namespace and finds the sleep package" { const blob = [_]u8{ // Name(_S5, Package(2){Byte 0x05, Byte 0x00}) 0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00, - // Scope(\_SB) pkglen=0x27 + // Scope(\_SB) packagelen=0x27 0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F, - // Device(PCI0) pkglen=0x1F + // Device(PCI0) packagelen=0x1F 0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30, // Name(_HID, 0x11) 0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11, - // Method(MTHD, flags=1) empty, pkglen=0x06 + // Method(MTHD, flags=1) empty, packagelen=0x06 0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01, - // Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B + // Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B 0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00, // OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1) 0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01, - // Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B + // Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B 0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08, }; @@ -149,32 +147,32 @@ test "parses a nested namespace and finds the sleep package" { try std.testing.expectEqual(blob.len, result.consumed); try std.testing.expectEqual(blob.len, result.total); - const ns = &result.namespace; + const namespace = &result.namespace; // Expected top-level nodes. - const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB; + const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB; try std.testing.expectEqual(NodeKind.scope, sb.kind); - const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0; + const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0; try std.testing.expectEqual(NodeKind.device, pci0.kind); - _ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID; + _ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID; // The 1-arg method's arg count was parsed from its flags byte. - const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD; + const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD; try std.testing.expectEqual(NodeKind.method, mthd.kind); try std.testing.expectEqual(@as(u8, 1), mthd.arg_count); // OperationRegion and the Field unit made it into the namespace. - _ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion; - _ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField; + _ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion; + _ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField; // The sleep package decoded. - const s5 = sleepState(ns, 5) orelse return error.NoS5; + const s5 = sleepState(namespace, 5) orelse return error.NoS5; try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a); try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b); } -fn noMap(phys: u64, _: u64, _: bool) u64 { - return phys; +fn noMap(physical: u64, _: u64, _: bool) u64 { + return physical; } fn noRead(_: u8, _: u16) u32 { return 0; @@ -198,13 +196,13 @@ test "interpreter runs a method with args, arithmetic, and control flow" { var arena = std.heap.ArenaAllocator.init(std.testing.allocator); defer arena.deinit(); var result = try parse(arena.allocator(), &.{&blob}); - const ns = &result.namespace; - const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod; + const namespace = &result.namespace; + const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod; - var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator()); + var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator()); - const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1 - try std.testing.expectEqual(@as(u64, 1), try hi.asInt()); - const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0 - try std.testing.expectEqual(@as(u64, 0), try lo.asInt()); + const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1 + try std.testing.expectEqual(@as(u64, 1), try hi.asInteger()); + const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0 + try std.testing.expectEqual(@as(u64, 0), try lo.asInteger()); } diff --git a/src/device/aml/interp.zig b/src/device/aml/interp.zig index 318baa8..bef89d8 100644 --- a/src/device/aml/interp.zig +++ b/src/device/aml/interp.zig @@ -14,14 +14,13 @@ //! fall back — never a hard failure. const std = @import("std"); -const op = @import("opcodes.zig"); -const nsp = @import("namespace.zig"); -const Node = nsp.Node; -const Namespace = nsp.Namespace; +const opcode = @import("opcodes.zig"); +const Node = @import("namespace.zig").Node; +const Namespace = @import("namespace.zig").Namespace; -/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio). +/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio). pub const Hal = struct { - mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64, + mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64, pioRead: *const fn (width: u8, port: u16) u32, pioWrite: *const fn (width: u8, port: u16, value: u32) void, }; @@ -37,7 +36,7 @@ pub const Object = union(enum) { package: []Object, reference: *Node, - pub fn asInt(self: Object) Error!u64 { + pub fn asInteger(self: Object) Error!u64 { return switch (self) { .integer => |v| v, .buffer => |b| blk: { @@ -53,14 +52,14 @@ pub const Object = union(enum) { } }; -const max_segs = 16; +const maximum_segments = 16; const NamePath = struct { rooted: bool = false, parents: u8 = 0, - segs: [max_segs][4]u8 = undefined, + segments: [maximum_segments][4]u8 = undefined, count: usize = 0, fn slice(self: *const NamePath) []const [4]u8 { - return self.segs[0..self.count]; + return self.segments[0..self.count]; } }; @@ -86,7 +85,7 @@ const Cursor = struct { self.i += n; return s; } - fn pkgLen(self: *Cursor) Error!usize { + fn packageLength(self: *Cursor) Error!usize { const lead = try self.byte(); const follow: usize = lead >> 6; if (follow == 0) return lead & 0x3F; @@ -96,36 +95,36 @@ const Cursor = struct { return value; } fn nameString(self: *Cursor) Error!NamePath { - var np = NamePath{}; - if (self.peek() == op.root_char) { - np.rooted = true; + var name_path = NamePath{}; + if (self.peek() == opcode.root_char) { + name_path.rooted = true; self.i += 1; } else { - while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1; + while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1; } - const lead = self.peek() orelse return np; + const lead = self.peek() orelse return name_path; switch (lead) { 0x00 => self.i += 1, - op.dual_name_prefix => { + opcode.dual_name_prefix => { self.i += 1; - try self.seg(&np); - try self.seg(&np); + try self.segment(&name_path); + try self.segment(&name_path); }, - op.multi_name_prefix => { + opcode.multi_name_prefix => { self.i += 1; - const cnt = try self.byte(); + const count = try self.byte(); var k: usize = 0; - while (k < cnt) : (k += 1) try self.seg(&np); + while (k < count) : (k += 1) try self.segment(&name_path); }, - else => try self.seg(&np), + else => try self.segment(&name_path), } - return np; + return name_path; } - fn seg(self: *Cursor, np: *NamePath) Error!void { + fn segment(self: *Cursor, name_path: *NamePath) Error!void { const s = try self.take(4); - if (np.count < max_segs) { - np.segs[np.count] = s[0..4].*; - np.count += 1; + if (name_path.count < maximum_segments) { + name_path.segments[name_path.count] = s[0..4].*; + name_path.count += 1; } } }; @@ -140,45 +139,45 @@ const Frame = struct { }; /// A CreateField binding: a name that indexes into a buffer object. -const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 }; +const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 }; -pub const Interp = struct { - ns: *Namespace, +pub const Interpreter = struct { + namespace: *Namespace, hal: Hal, arena: std.mem.Allocator, /// Runtime object overrides for Name nodes (Store targets, patched buffers). - dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{}, + dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{}, /// CreateField bindings active for the current evaluation. - fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{}, + fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{}, - pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp { - return .{ .ns = ns, .hal = hal, .arena = arena }; + pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter { + return .{ .namespace = namespace, .hal = hal, .arena = arena }; } /// Evaluate a namespace object: invoke a Method, read a Name's value, or read a /// Field. Resets per-evaluation runtime state first. - pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object { - self.dyn.clearRetainingCapacity(); + pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object { + self.dynamic_overrides.clearRetainingCapacity(); self.fields.clearRetainingCapacity(); return self.invoke(node, args); } - fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object { + fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object { switch (node.kind) { .method => { var frame = Frame{ .scope = node }; for (args, 0..) |a, i| { if (i < frame.args.len) frame.args[i] = a; } - var cur = Cursor{ .b = node.value }; - try self.execList(&cur, &frame); + var current = Cursor{ .b = node.value }; + try self.executeList(¤t, &frame); return frame.ret; }, .name => { - if (self.dyn.get(node)) |o| return o; - var cur = Cursor{ .b = node.value }; - var frame = Frame{ .scope = node.parent orelse self.ns.root }; - return self.term(&cur, &frame); + if (self.dynamic_overrides.get(node)) |o| return o; + var current = Cursor{ .b = node.value }; + var frame = Frame{ .scope = node.parent orelse self.namespace.root }; + return self.term(¤t, &frame); }, .field => return .{ .integer = try self.readField(node) }, else => return .{ .reference = node }, @@ -186,96 +185,96 @@ pub const Interp = struct { } /// Execute a TermList until it ends or the frame returns/breaks. - fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void { - while (!cur.eof() and !frame.returned and !frame.broke) { - _ = try self.term(cur, frame); + fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void { + while (!current.eof() and !frame.returned and !frame.broke) { + _ = try self.term(current, frame); } } /// Evaluate/execute one term, returning its value (`.uninitialized` for pure /// statements). - fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const lead = cur.peek() orelse return error.Truncated; - if (isNameStart(lead)) return self.nameRef(cur, frame); - _ = try cur.byte(); + fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const lead = current.peek() orelse return error.Truncated; + if (isNameStart(lead)) return self.nameReference(current, frame); + _ = try current.byte(); return switch (lead) { - op.zero_op => Object{ .integer = 0 }, - op.one_op => Object{ .integer = 1 }, - op.ones_op => Object{ .integer = ~@as(u64, 0) }, - op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) }, - op.word_prefix => Object{ .integer = try self.readConst(cur, 2) }, - op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) }, - op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) }, - op.string_prefix => try self.readString(cur), - op.buffer_op => try self.buffer(cur, frame), - op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op), + opcode.zero_opcode => Object{ .integer = 0 }, + opcode.one_opcode => Object{ .integer = 1 }, + opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) }, + opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) }, + opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) }, + opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) }, + opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) }, + opcode.string_prefix => try self.readString(current), + opcode.buffer_opcode => try self.buffer(current, frame), + opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode), - op.local0_op...op.local7_op => frame.locals[lead - op.local0_op], - op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op], + opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode], + opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode], - op.return_op => blk: { - frame.ret = try self.term(cur, frame); + opcode.return_opcode => blk: { + frame.ret = try self.term(current, frame); frame.returned = true; break :blk .uninitialized; }, - op.break_op => blk: { + opcode.break_opcode => blk: { frame.broke = true; break :blk .uninitialized; }, - op.continue_op, op.noop_op => .uninitialized, + opcode.continue_opcode, opcode.noop_opcode => .uninitialized, - op.if_op => try self.ifElse(cur, frame), - op.while_op => try self.whileLoop(cur, frame), - op.store_op => try self.store(cur, frame), - op.increment_op => try self.incDec(cur, frame, 1), - op.decrement_op => try self.incDec(cur, frame, -1), + opcode.if_opcode => try self.ifElse(current, frame), + opcode.while_opcode => try self.whileLoop(current, frame), + opcode.store_opcode => try self.store(current, frame), + opcode.increment_opcode => try self.incDec(current, frame, 1), + opcode.decrement_opcode => try self.incDec(current, frame, -1), - op.add_op => try self.binary(cur, frame, .add), - op.subtract_op => try self.binary(cur, frame, .sub), - op.multiply_op => try self.binary(cur, frame, .mul), - op.mod_op => try self.binary(cur, frame, .mod), - op.and_op => try self.binary(cur, frame, .band), - op.or_op => try self.binary(cur, frame, .bor), - op.xor_op => try self.binary(cur, frame, .bxor), - op.nand_op => try self.binary(cur, frame, .nand), - op.nor_op => try self.binary(cur, frame, .nor), - op.shift_left_op => try self.binary(cur, frame, .shl), - op.shift_right_op => try self.binary(cur, frame, .shr), - op.divide_op => try self.divide(cur, frame), + opcode.add_opcode => try self.binary(current, frame, .add), + opcode.subtract_opcode => try self.binary(current, frame, .sub), + opcode.multiply_opcode => try self.binary(current, frame, .mul), + opcode.mod_opcode => try self.binary(current, frame, .mod), + opcode.and_opcode => try self.binary(current, frame, .band), + opcode.or_opcode => try self.binary(current, frame, .bor), + opcode.xor_opcode => try self.binary(current, frame, .bxor), + opcode.nand_opcode => try self.binary(current, frame, .nand), + opcode.nor_opcode => try self.binary(current, frame, .nor), + opcode.shift_left_opcode => try self.binary(current, frame, .shl), + opcode.shift_right_opcode => try self.binary(current, frame, .shr), + opcode.divide_opcode => try self.divide(current, frame), - op.land_op => try self.logic2(cur, frame, .land), - op.lor_op => try self.logic2(cur, frame, .lor), - op.lequal_op => try self.logic2(cur, frame, .eq), - op.lgreater_op => try self.logic2(cur, frame, .gt), - op.lless_op => try self.logic2(cur, frame, .lt), - op.lnot_op => try self.lnot(cur, frame), + opcode.land_opcode => try self.logic2(current, frame, .land), + opcode.lor_opcode => try self.logic2(current, frame, .lor), + opcode.lequal_opcode => try self.logic2(current, frame, .eq), + opcode.lgreater_opcode => try self.logic2(current, frame, .gt), + opcode.lless_opcode => try self.logic2(current, frame, .lt), + opcode.lnot_opcode => try self.lnot(current, frame), - op.not_op => blk: { - const v = try self.evalInt(cur, frame); + opcode.not_opcode => blk: { + const v = try self.evaluateInteger(current, frame); const r = ~v; - try self.storeTarget(cur, frame, .{ .integer = r }); + try self.storeTarget(current, frame, .{ .integer = r }); break :blk .{ .integer = r }; }, - op.size_of_op => try self.sizeOf(cur, frame), - op.index_op => try self.index(cur, frame), - op.deref_of_op => try self.derefOf(cur, frame), - op.to_integer_op => blk: { - const v = try self.evalInt(cur, frame); - try self.storeTarget(cur, frame, .{ .integer = v }); + opcode.size_of_opcode => try self.sizeOf(current, frame), + opcode.index_opcode => try self.index(current, frame), + opcode.dereference_of_opcode => try self.dereferenceOf(current, frame), + opcode.to_integer_opcode => blk: { + const v = try self.evaluateInteger(current, frame); + try self.storeTarget(current, frame, .{ .integer = v }); break :blk .{ .integer = v }; }, - op.to_buffer_op => try self.passThroughUnary(cur, frame), + opcode.to_buffer_opcode => try self.passThroughUnary(current, frame), - op.ext_op_prefix => try self.ext(cur, frame), + opcode.extended_opcode_prefix => try self.ext(current, frame), // CreateXField: source, index, name (bit widths differ by op) - op.create_bit_field_op => try self.createField(cur, frame, 1), - op.create_byte_field_op => try self.createField(cur, frame, 8), - op.create_word_field_op => try self.createField(cur, frame, 16), - op.create_dword_field_op => try self.createField(cur, frame, 32), - op.create_qword_field_op => try self.createField(cur, frame, 64), + opcode.create_bit_field_opcode => try self.createField(current, frame, 1), + opcode.create_byte_field_opcode => try self.createField(current, frame, 8), + opcode.create_word_field_opcode => try self.createField(current, frame, 16), + opcode.create_dword_field_opcode => try self.createField(current, frame, 32), + opcode.create_qword_field_opcode => try self.createField(current, frame, 64), else => error.Unsupported, }; @@ -283,15 +282,15 @@ pub const Interp = struct { // --- name references ---------------------------------------------------- - fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const np = try cur.nameString(); - const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse + fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const name_path = try current.nameString(); + const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized; // unknown name -> treat as uninitialised switch (node.kind) { .method => { var argbuf: [7]Object = undefined; var i: usize = 0; - while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame); + while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame); return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]); }, .field => return .{ .integer = try self.readField(node) }, @@ -302,58 +301,58 @@ pub const Interp = struct { // --- data objects ------------------------------------------------------- - fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 { + fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 { _ = self; - const bytes = try cur.take(n); + const bytes = try current.take(n); var v: u64 = 0; for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8); return v; } - fn readString(self: *Interp, cur: *Cursor) Error!Object { - const start = cur.i; - while (cur.peek()) |c| { - cur.i += 1; + fn readString(self: *Interpreter, current: *Cursor) Error!Object { + const start = current.i; + while (current.peek()) |c| { + current.i += 1; if (c == 0) break; } - const raw = cur.b[start .. cur.i - 1]; + const raw = current.b[start .. current.i - 1]; const s = try self.arena.dupe(u8, raw); return .{ .string = s }; } - fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const start = cur.i; - const len = try cur.pkgLen(); - const end = @min(start + len, cur.b.len); - const size = try self.evalInt(cur, frame); - const data = cur.b[@min(cur.i, end)..end]; - const buf = try self.arena.alloc(u8, @intCast(size)); - @memset(buf, 0); - @memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]); - cur.i = end; - return .{ .buffer = buf }; + fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const start = current.i; + const len = try current.packageLength(); + const end = @min(start + len, current.b.len); + const size = try self.evaluateInteger(current, frame); + const data = current.b[@min(current.i, end)..end]; + const bytes = try self.arena.alloc(u8, @intCast(size)); + @memset(bytes, 0); + @memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]); + current.i = end; + return .{ .buffer = bytes }; } - fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object { - const start = cur.i; - const len = try cur.pkgLen(); - const end = @min(start + len, cur.b.len); - const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte(); + fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object { + const start = current.i; + const len = try current.packageLength(); + const end = @min(start + len, current.b.len); + const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte(); const elems = try self.arena.alloc(Object, count); var i: usize = 0; - while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame); + while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame); while (i < count) : (i += 1) elems[i] = .uninitialized; - cur.i = end; + current.i = end; return .{ .package = elems }; } // --- operators ---------------------------------------------------------- - const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr }; + const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr }; - fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object { - const a = try self.evalInt(cur, frame); - const b = try self.evalInt(cur, frame); + fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object { + const a = try self.evaluateInteger(current, frame); + const b = try self.evaluateInteger(current, frame); const r: u64 = switch (kind) { .add => a +% b, .sub => a -% b, @@ -367,24 +366,24 @@ pub const Interp = struct { .shl => if (b >= 64) 0 else a << @intCast(b), .shr => if (b >= 64) 0 else a >> @intCast(b), }; - try self.storeTarget(cur, frame, .{ .integer = r }); + try self.storeTarget(current, frame, .{ .integer = r }); return .{ .integer = r }; } - fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const a = try self.evalInt(cur, frame); - const b = try self.evalInt(cur, frame); + fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const a = try self.evaluateInteger(current, frame); + const b = try self.evaluateInteger(current, frame); if (b == 0) return error.DivByZero; - try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target - try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target + try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target + try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target return .{ .integer = a / b }; } - const LogicOp = enum { land, lor, eq, gt, lt }; + const LogicOperation = enum { land, lor, eq, gt, lt }; - fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object { - const a = try self.evalInt(cur, frame); - const b = try self.evalInt(cur, frame); + fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object { + const a = try self.evaluateInteger(current, frame); + const b = try self.evaluateInteger(current, frame); const r = switch (kind) { .land => a != 0 and b != 0, .lor => a != 0 or b != 0, @@ -395,48 +394,48 @@ pub const Interp = struct { return .{ .integer = if (r) ~@as(u64, 0) else 0 }; } - fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { + fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { // 0x92 0x93/94/95 are the compound comparisons. - const b = cur.peek() orelse return error.Truncated; + const b = current.peek() orelse return error.Truncated; switch (b) { - op.lnot.not_equal => { - cur.i += 1; - const x = try self.evalInt(cur, frame); - const y = try self.evalInt(cur, frame); + opcode.lnot.not_equal => { + current.i += 1; + const x = try self.evaluateInteger(current, frame); + const y = try self.evaluateInteger(current, frame); return .{ .integer = if (x != y) ~@as(u64, 0) else 0 }; }, - op.lnot.less_equal => { - cur.i += 1; - const x = try self.evalInt(cur, frame); - const y = try self.evalInt(cur, frame); + opcode.lnot.less_equal => { + current.i += 1; + const x = try self.evaluateInteger(current, frame); + const y = try self.evaluateInteger(current, frame); return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 }; }, - op.lnot.greater_equal => { - cur.i += 1; - const x = try self.evalInt(cur, frame); - const y = try self.evalInt(cur, frame); + opcode.lnot.greater_equal => { + current.i += 1; + const x = try self.evaluateInteger(current, frame); + const y = try self.evaluateInteger(current, frame); return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 }; }, else => { - const x = try self.evalInt(cur, frame); + const x = try self.evaluateInteger(current, frame); return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 }; }, } } - fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object { + fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object { // Operand is a SuperName that is both read and written. - const save = cur.i; - const cur_val = try self.term(cur, frame); - const v = try cur_val.asInt(); + const save = current.i; + const current_value = try self.term(current, frame); + const v = try current_value.asInteger(); const r = if (delta > 0) v +% 1 else v -% 1; - var tcur = Cursor{ .b = cur.b, .i = save }; + var tcur = Cursor{ .b = current.b, .i = save }; try self.storeInto(&tcur, frame, .{ .integer = r }); return .{ .integer = r }; } - fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const o = try self.term(cur, frame); + fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const o = try self.term(current, frame); return .{ .integer = switch (o) { .buffer => |b| b.len, .string => |s| s.len, @@ -445,29 +444,29 @@ pub const Interp = struct { } }; } - fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const o = try self.term(cur, frame); - try self.storeTarget(cur, frame, o); + fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const o = try self.term(current, frame); + try self.storeTarget(current, frame, o); return o; } - fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const src = try self.term(cur, frame); - const idx: usize = @intCast(try self.evalInt(cur, frame)); + fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const source = try self.term(current, frame); + const element_index: usize = @intCast(try self.evaluateInteger(current, frame)); // Optional target (a reference); we don't materialise references, so store // the indexed value if a target is present. - const val: Object = switch (src) { - .buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 }, - .package => |p| if (idx < p.len) p[idx] else .uninitialized, - .string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 }, + const value: Object = switch (source) { + .buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 }, + .package => |p| if (element_index < p.len) p[element_index] else .uninitialized, + .string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 }, else => .uninitialized, }; - try self.storeTarget(cur, frame, val); - return val; + try self.storeTarget(current, frame, value); + return value; } - fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const o = try self.term(cur, frame); + fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const o = try self.term(current, frame); return switch (o) { .reference => |n| self.invoke(n, &.{}), else => o, @@ -476,101 +475,101 @@ pub const Interp = struct { // --- control flow ------------------------------------------------------- - fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const start = cur.i; - const end = @min(start + try cur.pkgLen(), cur.b.len); - const cond = try self.evalInt(cur, frame); + fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const start = current.i; + const end = @min(start + try current.packageLength(), current.b.len); + const cond = try self.evaluateInteger(current, frame); if (cond != 0) { - var body = Cursor{ .b = cur.b[0..end], .i = cur.i }; - try self.execList(&body, frame); - cur.i = end; + var body = Cursor{ .b = current.b[0..end], .i = current.i }; + try self.executeList(&body, frame); + current.i = end; // Skip a trailing Else. - if (cur.peek() == op.else_op) { - cur.i += 1; - const es = cur.i; - const ee = @min(es + try cur.pkgLen(), cur.b.len); - cur.i = ee; + if (current.peek() == opcode.else_opcode) { + current.i += 1; + const es = current.i; + const ee = @min(es + try current.packageLength(), current.b.len); + current.i = ee; } } else { - cur.i = end; - if (cur.peek() == op.else_op) { - cur.i += 1; - const es = cur.i; - const ee = @min(es + try cur.pkgLen(), cur.b.len); - var body = Cursor{ .b = cur.b[0..ee], .i = cur.i }; - try self.execList(&body, frame); - cur.i = ee; + current.i = end; + if (current.peek() == opcode.else_opcode) { + current.i += 1; + const es = current.i; + const ee = @min(es + try current.packageLength(), current.b.len); + var body = Cursor{ .b = current.b[0..ee], .i = current.i }; + try self.executeList(&body, frame); + current.i = ee; } } return .uninitialized; } - fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const start = cur.i; - const end = @min(start + try cur.pkgLen(), cur.b.len); - const pred_at = cur.i; + fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const start = current.i; + const end = @min(start + try current.packageLength(), current.b.len); + const pred_at = current.i; var guard: usize = 0; while (guard < 100_000) : (guard += 1) { - var pc = Cursor{ .b = cur.b[0..end], .i = pred_at }; - const cond = try self.evalInt(&pc, frame); + var pc = Cursor{ .b = current.b[0..end], .i = pred_at }; + const cond = try self.evaluateInteger(&pc, frame); if (cond == 0) break; - var body = Cursor{ .b = cur.b[0..end], .i = pc.i }; - try self.execList(&body, frame); + var body = Cursor{ .b = current.b[0..end], .i = pc.i }; + try self.executeList(&body, frame); if (frame.returned) break; if (frame.broke) { frame.broke = false; break; } } - cur.i = end; + current.i = end; return .uninitialized; } // --- store -------------------------------------------------------------- - fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const value = try self.term(cur, frame); - try self.storeInto(cur, frame, value); + fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const value = try self.term(current, frame); + try self.storeInto(current, frame, value); return value; } /// A Store *target* that may be NullName (no store). - fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void { - if (cur.peek() == 0x00) { - cur.i += 1; // NullName + fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void { + if (current.peek() == 0x00) { + current.i += 1; // NullName return; } - try self.storeInto(cur, frame, value); + try self.storeInto(current, frame, value); } - fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void { - const lead = cur.peek() orelse return error.Truncated; + fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void { + const lead = current.peek() orelse return error.Truncated; if (isNameStart(lead)) { - const np = try cur.nameString(); - const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return; - if (self.fields.get(node)) |bf| { - try self.writeBufField(bf, try value.asInt()); + const name_path = try current.nameString(); + const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return; + if (self.fields.get(node)) |buffer_field| { + try self.writeBufferField(buffer_field, try value.asInteger()); } else if (node.kind == .field) { - try self.writeField(node, try value.asInt()); + try self.writeField(node, try value.asInteger()); } else { - try self.dyn.put(self.arena, node, value); + try self.dynamic_overrides.put(self.arena, node, value); } return; } - _ = try cur.byte(); + _ = try current.byte(); switch (lead) { 0x00 => {}, // NullName - op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value, - op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value, - op.index_op => { - const src = try self.term(cur, frame); - const idx: usize = @intCast(try self.evalInt(cur, frame)); - switch (src) { - .buffer => |b| if (idx < b.len) { - b[idx] = @truncate(try value.asInt()); + opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value, + opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value, + opcode.index_opcode => { + const source = try self.term(current, frame); + const element_index: usize = @intCast(try self.evaluateInteger(current, frame)); + switch (source) { + .buffer => |b| if (element_index < b.len) { + b[element_index] = @truncate(try value.asInteger()); }, - .package => |p| if (idx < p.len) { - p[idx] = value; + .package => |p| if (element_index < p.len) { + p[element_index] = value; }, else => {}, } @@ -581,144 +580,144 @@ pub const Interp = struct { // --- CreateField (buffer patching) -------------------------------------- - fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object { - const src = try self.term(cur, frame); // source buffer (as a reference or value) - const bit_index = try self.evalInt(cur, frame); - const np = try cur.nameString(); - const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized; + fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object { + const source = try self.term(current, frame); // source buffer (as a reference or value) + const bit_index = try self.evaluateInteger(current, frame); + const name_path = try current.nameString(); + const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized; // Bind the new name to the source buffer's node so stores land in it. - const buf_node: *Node = switch (src) { + const buffer_node: *Node = switch (source) { .reference => |n| n, else => return .uninitialized, }; - // Materialise the buffer into `dyn` so patches persist and are returned. - if (self.dyn.get(buf_node) == null) { - const val = try self.invoke(buf_node, &.{}); - try self.dyn.put(self.arena, buf_node, val); + // Materialise the buffer into `dynamic_overrides` so patches persist and are returned. + if (self.dynamic_overrides.get(buffer_node) == null) { + const value = try self.invoke(buffer_node, &.{}); + try self.dynamic_overrides.put(self.arena, buffer_node, value); } const byte_off: usize = @intCast(bit_index / 8); - try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width }); + try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width }); return .uninitialized; } - fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void { - const obj = self.dyn.get(bf.buf) orelse return; - const buf = switch (obj) { + fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void { + const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return; + const bytes = switch (obj) { .buffer => |b| b, else => return, }; - const nbytes = (bf.bit_width + 7) / 8; + const byte_count = (buffer_field.bit_width + 7) / 8; var k: usize = 0; - while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) { - buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8)); + while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) { + bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8)); } } // --- OperationRegion field access --------------------------------------- - fn readField(self: *Interp, field: *Node) Error!u64 { + fn readField(self: *Interpreter, field: *Node) Error!u64 { const region = field.region orelse return error.Unsupported; if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported; const base = try self.regionBase(region); const start_byte = base + field.bit_offset / 8; const shift: u7 = @intCast(field.bit_offset % 8); const total = @as(usize, shift) + field.bit_width; - const nbytes = (total + 7) / 8; + const byte_count = (total + 7) / 8; var raw: u128 = 0; var k: usize = 0; - while (k < nbytes) : (k += 1) { + while (k < byte_count) : (k += 1) { raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8); } const masked = (raw >> shift) & bitMask(field.bit_width); return @truncate(masked); } - fn writeField(self: *Interp, field: *Node, value: u64) Error!void { + fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void { const region = field.region orelse return error.Unsupported; if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported; const base = try self.regionBase(region); const start_byte = base + field.bit_offset / 8; const shift: u7 = @intCast(field.bit_offset % 8); const total = @as(usize, shift) + field.bit_width; - const nbytes = (total + 7) / 8; + const byte_count = (total + 7) / 8; // Read-modify-write byte by byte. var raw: u128 = 0; var k: usize = 0; - while (k < nbytes) : (k += 1) { + while (k < byte_count) : (k += 1) { raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8); } const mask = bitMask(field.bit_width) << shift; raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask); k = 0; - while (k < nbytes) : (k += 1) { + while (k < byte_count) : (k += 1) { try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8))); } } - fn regionBase(self: *Interp, region: *Node) Error!u64 { - var cur = Cursor{ .b = region.region_offset_aml }; - var frame = Frame{ .scope = region.parent orelse self.ns.root }; - return (try self.term(&cur, &frame)).asInt(); + fn regionBase(self: *Interpreter, region: *Node) Error!u64 { + var current = Cursor{ .b = region.region_offset_aml }; + var frame = Frame{ .scope = region.parent orelse self.namespace.root }; + return (try self.term(¤t, &frame)).asInteger(); } - fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 { + fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 { switch (space) { 0 => { // SystemMemory - const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true); - const p: *align(1) const volatile u8 = @ptrFromInt(virt + (addr & 0xFFF)); + const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true); + const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF)); return p.*; }, - 1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO + 1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO else => return error.Unsupported, } } - fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void { + fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void { switch (space) { 0 => { - const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true); - const p: *align(1) volatile u8 = @ptrFromInt(virt + (addr & 0xFFF)); + const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true); + const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF)); p.* = value; }, - 1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value), + 1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value), else => return error.Unsupported, } } // --- extended opcodes --------------------------------------------------- - fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object { - const e = try cur.byte(); + fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object { + const e = try current.byte(); switch (e) { - op.ext.debug => return .uninitialized, - op.ext.revision => return .{ .integer = 2 }, - op.ext.timer => return .{ .integer = 0 }, + opcode.extended.debug => return .uninitialized, + opcode.extended.revision => return .{ .integer = 2 }, + opcode.extended.timer => return .{ .integer = 0 }, // Mutex/Event ops are no-ops in this single-threaded evaluator. - op.ext.acquire => { - _ = try self.term(cur, frame); // mutex SuperName - _ = try cur.take(2); // timeout + opcode.extended.acquire => { + _ = try self.term(current, frame); // mutex SuperName + _ = try current.take(2); // timeout return .{ .integer = 0 }; // acquired }, - op.ext.release, op.ext.reset, op.ext.signal => { - _ = try self.term(cur, frame); + opcode.extended.release, opcode.extended.reset, opcode.extended.signal => { + _ = try self.term(current, frame); return .uninitialized; }, - op.ext.wait => { - _ = try self.term(cur, frame); - _ = try self.term(cur, frame); + opcode.extended.wait => { + _ = try self.term(current, frame); + _ = try self.term(current, frame); return .{ .integer = 0 }; }, - op.ext.sleep, op.ext.stall => { - _ = try self.term(cur, frame); + opcode.extended.sleep, opcode.extended.stall => { + _ = try self.term(current, frame); return .uninitialized; }, else => return error.Unsupported, } } - fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 { - return (try self.term(cur, frame)).asInt(); + fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 { + return (try self.term(current, frame)).asInteger(); } }; @@ -728,10 +727,10 @@ fn bitMask(width: u32) u128 { } fn isNameStart(b: u8) bool { - return (b >= op.name_char_start and b <= op.name_char_end) or - b == op.name_char_underscore or - b == op.root_char or - b == op.parent_prefix_char or - b == op.dual_name_prefix or - b == op.multi_name_prefix; + return (b >= opcode.name_char_start and b <= opcode.name_char_end) or + b == opcode.name_char_underscore or + b == opcode.root_char or + b == opcode.parent_prefix_char or + b == opcode.dual_name_prefix or + b == opcode.multi_name_prefix; } diff --git a/src/device/aml/namespace.zig b/src/device/aml/namespace.zig index ea0ed76..2182751 100644 --- a/src/device/aml/namespace.zig +++ b/src/device/aml/namespace.zig @@ -18,7 +18,7 @@ pub const NodeKind = enum { mutex, event, processor, - power_res, + power_resource, thermal_zone, alias, external, @@ -28,12 +28,12 @@ pub const NodeKind = enum { pub const Node = struct { /// The 4-byte NameSeg identifying this node within its parent. The root uses /// all-zero. - seg: [4]u8 = .{ 0, 0, 0, 0 }, + segment: [4]u8 = .{ 0, 0, 0, 0 }, kind: NodeKind = .other, /// For Method / External: the declared argument count (0..7). Used to resolve /// how many TermArgs a method invocation consumes. arg_count: u8 = 0, - /// For Name: the AML bytes of its DataRefObject (so a value like a sleep + /// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep /// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body, /// interpreted on demand by the evaluator. Empty otherwise. value: []const u8 = &.{}, @@ -78,49 +78,49 @@ pub const Namespace = struct { return self.root.subtreeCount(); } - fn findChild(parent: *Node, seg: [4]u8) ?*Node { + fn findChild(parent: *Node, segment: [4]u8) ?*Node { var c = parent.first_child; while (c) |child| : (c = child.next_sibling) { - if (std.mem.eql(u8, &child.seg, &seg)) return child; + if (std.mem.eql(u8, &child.segment, &segment)) return child; } return null; } - /// The direct child of `node` named `seg`, or null. Unlike `resolve`, this does + /// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does /// not apply the search-rule walk-up — it looks only at immediate children (for /// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)). - pub fn childOf(node: *Node, seg: [4]u8) ?*Node { - return findChild(node, seg); + pub fn childOf(node: *Node, segment: [4]u8) ?*Node { + return findChild(node, segment); } - fn newChild(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node { + fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node { const n = try self.allocator.create(Node); - n.* = .{ .seg = seg, .kind = kind, .parent = parent }; + n.* = .{ .segment = segment, .kind = kind, .parent = parent }; // Append at the tail so a dump reads in declaration order. if (parent.first_child == null) { parent.first_child = n; } else { - var cur = parent.first_child.?; - while (cur.next_sibling) |sib| cur = sib; - cur.next_sibling = n; + var current = parent.first_child.?; + while (current.next_sibling) |sib| current = sib; + current.next_sibling = n; } return n; } /// Create a Field unit node directly under `scope` (field units live in the /// scope of the Field/IndexField/BankField, not under the region). - pub fn newFieldUnit(self: *Namespace, scope: *Node, seg: [4]u8) !*Node { - return self.findOrCreate(scope, seg, .field); + pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node { + return self.findOrCreate(scope, segment, .field); } - fn findOrCreate(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node { - if (findChild(parent, seg)) |existing| { + fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node { + if (findChild(parent, segment)) |existing| { // Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more // specific kind rather than downgrading to a plain scope. if (existing.kind == .scope and kind != .scope) existing.kind = kind; return existing; } - return self.newChild(parent, seg, kind); + return self.newChild(parent, segment, kind); } /// The node a definition's NameString names, creating any intermediate scopes. @@ -131,16 +131,16 @@ pub const Namespace = struct { current: *Node, rooted: bool, parents: u8, - segs: []const [4]u8, + segments: []const [4]u8, kind: NodeKind, ) !*Node { var base = startNode(self, current, rooted, parents); - if (segs.len == 0) return base; + if (segments.len == 0) return base; var i: usize = 0; - while (i + 1 < segs.len) : (i += 1) { - base = try self.findOrCreate(base, segs[i], .scope); + while (i + 1 < segments.len) : (i += 1) { + base = try self.findOrCreate(base, segments[i], .scope); } - return self.findOrCreate(base, segs[segs.len - 1], kind); + return self.findOrCreate(base, segments[segments.len - 1], kind); } /// Resolve a NameString *reference* to an existing node, or null. A single @@ -151,22 +151,22 @@ pub const Namespace = struct { current: *Node, rooted: bool, parents: u8, - segs: []const [4]u8, + segments: []const [4]u8, ) ?*Node { - if (segs.len == 0) return null; + if (segments.len == 0) return null; - if (!rooted and parents == 0 and segs.len == 1) { + if (!rooted and parents == 0 and segments.len == 1) { // Search rule: this scope, then each ancestor up to the root. var scope: ?*Node = current; while (scope) |s| : (scope = s.parent) { - if (findChild(s, segs[0])) |n| return n; + if (findChild(s, segments[0])) |n| return n; } return null; } var base = startNode(self, current, rooted, parents); - for (segs) |seg| { - base = findChild(base, seg) orelse return null; + for (segments) |segment| { + base = findChild(base, segment) orelse return null; } return base; } diff --git a/src/device/aml/opcodes.zig b/src/device/aml/opcodes.zig index eb894dd..e97121c 100644 --- a/src/device/aml/opcodes.zig +++ b/src/device/aml/opcodes.zig @@ -2,28 +2,28 @@ //! //! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence //! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison -//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`). +//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`). // --- name / path characters ------------------------------------------------- -pub const zero_op = 0x00; -pub const one_op = 0x01; -pub const alias_op = 0x06; -pub const name_op = 0x08; +pub const zero_opcode = 0x00; +pub const one_opcode = 0x01; +pub const alias_opcode = 0x06; +pub const name_opcode = 0x08; pub const byte_prefix = 0x0A; pub const word_prefix = 0x0B; pub const dword_prefix = 0x0C; pub const string_prefix = 0x0D; pub const qword_prefix = 0x0E; -pub const scope_op = 0x10; -pub const buffer_op = 0x11; -pub const package_op = 0x12; -pub const var_package_op = 0x13; -pub const method_op = 0x14; -pub const external_op = 0x15; +pub const scope_opcode = 0x10; +pub const buffer_opcode = 0x11; +pub const package_opcode = 0x12; +pub const var_package_opcode = 0x13; +pub const method_opcode = 0x14; +pub const external_opcode = 0x15; pub const dual_name_prefix = 0x2E; pub const multi_name_prefix = 0x2F; -pub const ext_op_prefix = 0x5B; +pub const extended_opcode_prefix = 0x5B; pub const root_char = 0x5C; pub const parent_prefix_char = 0x5E; pub const name_char_underscore = 0x5F; @@ -34,80 +34,80 @@ pub const name_char_start = 0x41; // 'A' pub const name_char_end = 0x5A; // 'Z' // --- locals / args ---------------------------------------------------------- -pub const local0_op = 0x60; -pub const local7_op = 0x67; -pub const arg0_op = 0x68; -pub const arg6_op = 0x6E; +pub const local0_opcode = 0x60; +pub const local7_opcode = 0x67; +pub const arg0_opcode = 0x68; +pub const arg6_opcode = 0x6E; // --- store / references / arithmetic --------------------------------------- -pub const store_op = 0x70; -pub const ref_of_op = 0x71; -pub const add_op = 0x72; -pub const concat_op = 0x73; -pub const subtract_op = 0x74; -pub const increment_op = 0x75; -pub const decrement_op = 0x76; -pub const multiply_op = 0x77; -pub const divide_op = 0x78; -pub const shift_left_op = 0x79; -pub const shift_right_op = 0x7A; -pub const and_op = 0x7B; -pub const nand_op = 0x7C; -pub const or_op = 0x7D; -pub const nor_op = 0x7E; -pub const xor_op = 0x7F; -pub const not_op = 0x80; -pub const find_set_left_bit_op = 0x81; -pub const find_set_right_bit_op = 0x82; -pub const deref_of_op = 0x83; -pub const concat_res_op = 0x84; -pub const mod_op = 0x85; -pub const notify_op = 0x86; -pub const size_of_op = 0x87; -pub const index_op = 0x88; -pub const match_op = 0x89; -pub const create_dword_field_op = 0x8A; -pub const create_word_field_op = 0x8B; -pub const create_byte_field_op = 0x8C; -pub const create_bit_field_op = 0x8D; -pub const object_type_op = 0x8E; -pub const create_qword_field_op = 0x8F; +pub const store_opcode = 0x70; +pub const ref_of_opcode = 0x71; +pub const add_opcode = 0x72; +pub const concat_opcode = 0x73; +pub const subtract_opcode = 0x74; +pub const increment_opcode = 0x75; +pub const decrement_opcode = 0x76; +pub const multiply_opcode = 0x77; +pub const divide_opcode = 0x78; +pub const shift_left_opcode = 0x79; +pub const shift_right_opcode = 0x7A; +pub const and_opcode = 0x7B; +pub const nand_opcode = 0x7C; +pub const or_opcode = 0x7D; +pub const nor_opcode = 0x7E; +pub const xor_opcode = 0x7F; +pub const not_opcode = 0x80; +pub const find_set_left_bit_opcode = 0x81; +pub const find_set_right_bit_opcode = 0x82; +pub const dereference_of_opcode = 0x83; +pub const concat_resource_opcode = 0x84; +pub const mod_opcode = 0x85; +pub const notify_opcode = 0x86; +pub const size_of_opcode = 0x87; +pub const index_opcode = 0x88; +pub const match_opcode = 0x89; +pub const create_dword_field_opcode = 0x8A; +pub const create_word_field_opcode = 0x8B; +pub const create_byte_field_opcode = 0x8C; +pub const create_bit_field_opcode = 0x8D; +pub const object_type_opcode = 0x8E; +pub const create_qword_field_opcode = 0x8F; -pub const land_op = 0x90; -pub const lor_op = 0x91; -pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`) -pub const lequal_op = 0x93; -pub const lgreater_op = 0x94; -pub const lless_op = 0x95; -pub const to_buffer_op = 0x96; -pub const to_decimal_string_op = 0x97; -pub const to_hex_string_op = 0x98; -pub const to_integer_op = 0x99; -pub const to_string_op = 0x9C; -pub const copy_object_op = 0x9D; -pub const mid_op = 0x9E; -pub const continue_op = 0x9F; -pub const if_op = 0xA0; -pub const else_op = 0xA1; -pub const while_op = 0xA2; -pub const noop_op = 0xA3; -pub const return_op = 0xA4; -pub const break_op = 0xA5; -pub const break_point_op = 0xCC; -pub const ones_op = 0xFF; +pub const land_opcode = 0x90; +pub const lor_opcode = 0x91; +pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`) +pub const lequal_opcode = 0x93; +pub const lgreater_opcode = 0x94; +pub const lless_opcode = 0x95; +pub const to_buffer_opcode = 0x96; +pub const to_decimal_string_opcode = 0x97; +pub const to_hex_string_opcode = 0x98; +pub const to_integer_opcode = 0x99; +pub const to_string_opcode = 0x9C; +pub const copy_object_opcode = 0x9D; +pub const mid_opcode = 0x9E; +pub const continue_opcode = 0x9F; +pub const if_opcode = 0xA0; +pub const else_opcode = 0xA1; +pub const while_opcode = 0xA2; +pub const noop_opcode = 0xA3; +pub const return_opcode = 0xA4; +pub const break_opcode = 0xA5; +pub const break_point_opcode = 0xCC; +pub const ones_opcode = 0xFF; -/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes. +/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes. pub const lnot = struct { pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93 pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94 pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95 }; -/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B). -pub const ext = struct { +/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B). +pub const extended = struct { pub const mutex = 0x01; pub const event = 0x02; - pub const cond_ref_of = 0x12; + pub const conditional_reference_of = 0x12; pub const create_field = 0x13; pub const load_table = 0x1F; pub const load = 0x20; @@ -125,11 +125,11 @@ pub const ext = struct { pub const debug = 0x31; pub const fatal = 0x32; pub const timer = 0x33; - pub const op_region = 0x80; + pub const operation_region = 0x80; pub const field = 0x81; pub const device = 0x82; pub const processor = 0x83; - pub const power_res = 0x84; + pub const power_resource = 0x84; pub const thermal_zone = 0x85; pub const index_field = 0x86; pub const bank_field = 0x87; diff --git a/src/device/aml/parser.zig b/src/device/aml/parser.zig index 6d16271..56817cb 100644 --- a/src/device/aml/parser.zig +++ b/src/device/aml/parser.zig @@ -14,31 +14,31 @@ //! *within* one such object; the enclosing walk realigns at the boundary. const std = @import("std"); -const op = @import("opcodes.zig"); -const ns = @import("namespace.zig"); -const Namespace = ns.Namespace; -const Node = ns.Node; +const opcode = @import("opcodes.zig"); +const Namespace = @import("namespace.zig").Namespace; +const Node = @import("namespace.zig").Node; +const NodeKind = @import("namespace.zig").NodeKind; pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error; -const max_segs = 64; +const maximum_segments = 64; /// A parsed NameString: an optional root anchor or some parent hops, then a list /// of 4-byte segments. const NamePath = struct { rooted: bool = false, parents: u8 = 0, - segs: [max_segs][4]u8 = undefined, + segments: [maximum_segments][4]u8 = undefined, count: usize = 0, fn slice(self: *const NamePath) []const [4]u8 { - return self.segs[0..self.count]; + return self.segments[0..self.count]; } }; pub const Parser = struct { aml: []const u8, - pos: usize = 0, + position: usize = 0, namespace: *Namespace, pub fn init(aml: []const u8, namespace: *Namespace) Parser { @@ -49,29 +49,29 @@ pub const Parser = struct { /// number of bytes consumed — equal to `aml.len` for a clean full traversal. pub fn parseAll(self: *Parser) usize { self.termList(self.aml.len, self.namespace.root); - return self.pos; + return self.position; } // --- cursor primitives -------------------------------------------------- fn eof(self: *Parser) bool { - return self.pos >= self.aml.len; + return self.position >= self.aml.len; } fn peek(self: *Parser) ?u8 { - return if (self.eof()) null else self.aml[self.pos]; + return if (self.eof()) null else self.aml[self.position]; } fn readByte(self: *Parser) Error!u8 { if (self.eof()) return error.Truncated; - const b = self.aml[self.pos]; - self.pos += 1; + const b = self.aml[self.position]; + self.position += 1; return b; } fn skip(self: *Parser, n: usize) Error!void { - if (self.pos + n > self.aml.len) return error.Truncated; - self.pos += n; + if (self.position + n > self.aml.len) return error.Truncated; + self.position += n; } fn skipCString(self: *Parser) Error!void { @@ -83,7 +83,7 @@ pub const Parser = struct { /// AML PkgLength: the lead byte's top two bits give how many extra bytes /// follow; the value counts from the start of the PkgLength field. - fn readPkgLength(self: *Parser) Error!usize { + fn readPackageLength(self: *Parser) Error!usize { const lead = try self.readByte(); const follow: usize = lead >> 6; if (follow == 0) return lead & 0x3F; @@ -96,49 +96,49 @@ pub const Parser = struct { return value; } - fn readNameSeg(self: *Parser) Error![4]u8 { - if (self.pos + 4 > self.aml.len) return error.Truncated; - const seg = self.aml[self.pos..][0..4].*; - self.pos += 4; - return seg; + fn readNameSegment(self: *Parser) Error![4]u8 { + if (self.position + 4 > self.aml.len) return error.Truncated; + const segment = self.aml[self.position..][0..4].*; + self.position += 4; + return segment; } fn readNameString(self: *Parser) Error!NamePath { - var np = NamePath{}; + var name_path = NamePath{}; // A NameString is either root-anchored or parent-relative, not both. - if (self.peek() == op.root_char) { - np.rooted = true; - self.pos += 1; + if (self.peek() == opcode.root_char) { + name_path.rooted = true; + self.position += 1; } else { - while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1; + while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1; } - const lead = self.peek() orelse return np; + const lead = self.peek() orelse return name_path; switch (lead) { - 0x00 => self.pos += 1, // NullName - op.dual_name_prefix => { - self.pos += 1; - try self.appendSeg(&np); - try self.appendSeg(&np); + 0x00 => self.position += 1, // NullName + opcode.dual_name_prefix => { + self.position += 1; + try self.appendSegment(&name_path); + try self.appendSegment(&name_path); }, - op.multi_name_prefix => { - self.pos += 1; - const cnt = try self.readByte(); + opcode.multi_name_prefix => { + self.position += 1; + const count = try self.readByte(); var i: usize = 0; - while (i < cnt) : (i += 1) try self.appendSeg(&np); + while (i < count) : (i += 1) try self.appendSegment(&name_path); }, else => { - if (isNameStart(lead)) try self.appendSeg(&np); + if (isNameStart(lead)) try self.appendSegment(&name_path); }, } - return np; + return name_path; } - fn appendSeg(self: *Parser, np: *NamePath) Error!void { - const seg = try self.readNameSeg(); - if (np.count < max_segs) { - np.segs[np.count] = seg; - np.count += 1; + fn appendSegment(self: *Parser, name_path: *NamePath) Error!void { + const segment = try self.readNameSegment(); + if (name_path.count < maximum_segments) { + name_path.segments[name_path.count] = segment; + name_path.count += 1; } } @@ -147,10 +147,10 @@ pub const Parser = struct { /// Parse objects until `end`, then snap to `end`. Any parse error resyncs to /// the boundary rather than propagating — containment for the rare desync. fn termList(self: *Parser, end: usize, scope: *Node) void { - while (self.pos < end) { + while (self.position < end) { self.object(scope) catch break; } - self.pos = end; + self.position = end; } /// Parse exactly one object/term at the cursor. Used for both TermObjs and @@ -163,66 +163,66 @@ pub const Parser = struct { _ = try self.readByte(); switch (lead) { // constants and no-operand statements - op.zero_op, op.one_op, op.ones_op => {}, - op.noop_op, op.continue_op, op.break_op, op.break_point_op => {}, - op.local0_op...op.local7_op => {}, - op.arg0_op...op.arg6_op => {}, + opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {}, + opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {}, + opcode.local0_opcode...opcode.local7_opcode => {}, + opcode.arg0_opcode...opcode.arg6_opcode => {}, // literal data - op.byte_prefix => try self.skip(1), - op.word_prefix => try self.skip(2), - op.dword_prefix => try self.skip(4), - op.qword_prefix => try self.skip(8), - op.string_prefix => try self.skipCString(), + opcode.byte_prefix => try self.skip(1), + opcode.word_prefix => try self.skip(2), + opcode.dword_prefix => try self.skip(4), + opcode.qword_prefix => try self.skip(8), + opcode.string_prefix => try self.skipCString(), // data containers (contents skipped via their PkgLength) - op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(), + opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(), // namespace modifiers / named objects - op.name_op => try self.opName(scope), - op.alias_op => try self.opAlias(scope), - op.scope_op => try self.opScopeLike(scope, .scope), - op.method_op => try self.opMethod(scope), - op.external_op => try self.opExternal(scope), - op.ext_op_prefix => try self.opExt(scope), + opcode.name_opcode => try self.parseName(scope), + opcode.alias_opcode => try self.parseAlias(scope), + opcode.scope_opcode => try self.parseScopeLike(scope, .scope), + opcode.method_opcode => try self.parseMethod(scope), + opcode.external_opcode => try self.parseExternal(scope), + opcode.extended_opcode_prefix => try self.parseExtended(scope), // control flow - op.if_op => try self.opIf(scope), - op.else_op => try self.opElse(scope), - op.while_op => try self.opWhile(scope), - op.return_op => try self.object(scope), - op.notify_op => try self.args(scope, 2), + opcode.if_opcode => try self.parseIf(scope), + opcode.else_opcode => try self.parseElse(scope), + opcode.while_opcode => try self.parseWhile(scope), + opcode.return_opcode => try self.object(scope), + opcode.notify_opcode => try self.args(scope, 2), // stores / references / unary+target - op.store_op => try self.args(scope, 2), - op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1), - op.increment_op, op.decrement_op => try self.args(scope, 1), - op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2), - op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2), - op.copy_object_op => try self.args(scope, 2), + opcode.store_opcode => try self.args(scope, 2), + opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1), + opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1), + opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2), + opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2), + opcode.copy_object_opcode => try self.args(scope, 2), // binary + target - op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3), - op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3), - op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3), - op.divide_op => try self.args(scope, 4), - op.to_string_op => try self.args(scope, 3), - op.mid_op => try self.args(scope, 4), + opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3), + opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3), + opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3), + opcode.divide_opcode => try self.args(scope, 4), + opcode.to_string_opcode => try self.args(scope, 3), + opcode.mid_opcode => try self.args(scope, 4), // logical - op.land_op, op.lor_op => try self.args(scope, 2), - op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2), - op.lnot_op => try self.opLnot(scope), + opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2), + opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2), + opcode.lnot_opcode => try self.parseLnot(scope), - op.match_op => try self.opMatch(scope), + opcode.match_opcode => try self.parseMatch(scope), // CreateXField: NameString - op.create_dword_field_op, - op.create_word_field_op, - op.create_byte_field_op, - op.create_bit_field_op, - op.create_qword_field_op, - => try self.opCreateField(scope, 2), + opcode.create_dword_field_opcode, + opcode.create_word_field_opcode, + opcode.create_byte_field_opcode, + opcode.create_bit_field_opcode, + opcode.create_qword_field_opcode, + => try self.parseCreateField(scope, 2), else => return error.Malformed, } @@ -237,8 +237,8 @@ pub const Parser = struct { /// A NameString in operand/statement position: a method invocation (consuming /// the callee's declared argument count) or a plain name reference. fn nameInvocation(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); - if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| { + const name_path = try self.readNameString(); + if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| { if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) { try self.args(scope, node.arg_count); } @@ -247,132 +247,132 @@ pub const Parser = struct { /// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage): /// the contents are pure data, never namespace declarations. - fn skipPkg(self: *Parser) Error!void { - const start = self.pos; - const len = try self.readPkgLength(); + fn skipPackage(self: *Parser) Error!void { + const start = self.position; + const len = try self.readPackageLength(); const end = start + len; if (end > self.aml.len) return error.Truncated; - self.pos = end; + self.position = end; } // --- namespace objects -------------------------------------------------- - fn opName(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); - const val_start = self.pos; - try self.object(scope); // the DataRefObject value - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name); - node.value = self.aml[val_start..self.pos]; + fn parseName(self: *Parser, scope: *Node) Error!void { + const name_path = try self.readNameString(); + const value_start = self.position; + try self.object(scope); // the DataReferenceObject value + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name); + node.value = self.aml[value_start..self.position]; } - fn opAlias(self: *Parser, scope: *Node) Error!void { + fn parseAlias(self: *Parser, scope: *Node) Error!void { _ = try self.readNameString(); // source - const np = try self.readNameString(); // the alias name - _ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias); + const name_path = try self.readNameString(); // the alias name + _ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias); } - fn opMethod(self: *Parser, scope: *Node) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); - const np = try self.readNameString(); + fn parseMethod(self: *Parser, scope: *Node) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); + const name_path = try self.readNameString(); const flags = try self.readByte(); - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method); + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method); node.arg_count = flags & 0x7; // Capture the body for on-demand evaluation and skip it — objects declared // inside a method are created at *runtime*, not at load, so they must not // become permanent namespace nodes. - node.value = self.aml[self.pos..@min(end, self.aml.len)]; - self.pos = end; + node.value = self.aml[self.position..@min(end, self.aml.len)]; + self.position = end; } - fn opExternal(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); + fn parseExternal(self: *Parser, scope: *Node) Error!void { + const name_path = try self.readNameString(); _ = try self.readByte(); // object type const arg_count = try self.readByte(); - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external); + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external); node.arg_count = arg_count; } /// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList. - fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); - const np = try self.readNameString(); - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind); + fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); + const name_path = try self.readNameString(); + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind); self.termList(end, node); } - fn opProcessor(self: *Parser, scope: *Node) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); - const np = try self.readNameString(); - try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte) - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor); + fn parseProcessor(self: *Parser, scope: *Node) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); + const name_path = try self.readNameString(); + try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte) + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor); self.termList(end, node); } - fn opPowerRes(self: *Parser, scope: *Node) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); - const np = try self.readNameString(); + fn parsePowerResource(self: *Parser, scope: *Node) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); + const name_path = try self.readNameString(); try self.skip(3); // SystemLevel(byte) + ResourceOrder(word) - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res); + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource); self.termList(end, node); } /// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg). /// The offset/length expressions are kept as AML for lazy evaluation. - fn opRegion(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); + fn parseRegion(self: *Parser, scope: *Node) Error!void { + const name_path = try self.readNameString(); const space = try self.readByte(); - const off_start = self.pos; + const off_start = self.position; try self.object(scope); - const off_end = self.pos; + const off_end = self.position; try self.object(scope); - const len_end = self.pos; - const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region); + const len_end = self.position; + const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region); node.region_space = space; node.region_offset_aml = self.aml[off_start..off_end]; node.region_len_aml = self.aml[off_end..len_end]; } - fn opDataRegion(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); + fn parseDataRegion(self: *Parser, scope: *Node) Error!void { + const name_path = try self.readNameString(); try self.args(scope, 3); // signature, oem id, oem table id (TermArgs) - _ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region); + _ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region); } - fn opMutex(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); + fn parseMutex(self: *Parser, scope: *Node) Error!void { + const name_path = try self.readNameString(); try self.skip(1); // sync flags - _ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex); + _ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex); } - fn opEvent(self: *Parser, scope: *Node) Error!void { - const np = try self.readNameString(); - _ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event); + fn parseEvent(self: *Parser, scope: *Node) Error!void { + const name_path = try self.readNameString(); + _ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event); } /// CreateXField: `count` TermArgs then the new field's NameString. - fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void { + fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void { try self.args(scope, count); - const np = try self.readNameString(); - _ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name); + const name_path = try self.readNameString(); + _ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name); } /// Field / IndexField / BankField: a region/bank reference, flags, then a /// FieldList whose NamedFields become nodes in the current scope. For a plain /// Field, the first NameString is the backing region — captured so field units /// carry a region + bit position the evaluator can read/write. - fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); + fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); var region: ?*Node = null; var i: u8 = 0; while (i < name_strings) : (i += 1) { - const np = try self.readNameString(); + const name_path = try self.readNameString(); // Only a plain Field's single NameString denotes an OperationRegion. - if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice()); + if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice()); } if (bank) try self.object(scope); // bank value TermArg const flags = try self.readByte(); @@ -382,32 +382,32 @@ pub const Parser = struct { fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void { var bit_offset: u32 = 0; var access = initial_access; - while (self.pos < end) { + while (self.position < end) { const lead = self.peek() orelse break; switch (lead) { 0x00 => { // ReservedField: advances the bit position - self.pos += 1; - const width = self.readPkgLength() catch break; + self.position += 1; + const width = self.readPackageLength() catch break; bit_offset += @intCast(width); }, 0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib - self.pos += 1; + self.position += 1; const at = self.readByte() catch break; self.skip(1) catch break; access = at & 0x0F; }, 0x02 => { // ConnectField: NameString | BufferData - self.pos += 1; + self.position += 1; self.object(scope) catch break; }, 0x03 => { // ExtendedAccessField: type + attrib + length - self.pos += 1; + self.position += 1; self.skip(3) catch break; }, - else => { // NamedField: NameSeg + PkgLength (bit width) - const seg = self.readNameSeg() catch break; - const width = self.readPkgLength() catch break; - const unit = self.namespace.newFieldUnit(scope, seg) catch break; + else => { // NamedField: NameSegment + PkgLength (bit width) + const segment = self.readNameSegment() catch break; + const width = self.readPackageLength() catch break; + const unit = self.namespace.newFieldUnit(scope, segment) catch break; unit.region = region; unit.bit_offset = bit_offset; unit.bit_width = @intCast(width); @@ -416,49 +416,49 @@ pub const Parser = struct { }, } } - self.pos = end; + self.position = end; } // --- control flow ------------------------------------------------------- - fn opIf(self: *Parser, scope: *Node) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); + fn parseIf(self: *Parser, scope: *Node) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); try self.object(scope); // predicate self.termList(end, scope); - if (self.peek() == op.else_op) { - self.pos += 1; - try self.opElse(scope); + if (self.peek() == opcode.else_opcode) { + self.position += 1; + try self.parseElse(scope); } } - fn opElse(self: *Parser, scope: *Node) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); + fn parseElse(self: *Parser, scope: *Node) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); self.termList(end, scope); } - fn opWhile(self: *Parser, scope: *Node) Error!void { - const start = self.pos; - const end = start + try self.readPkgLength(); + fn parseWhile(self: *Parser, scope: *Node) Error!void { + const start = self.position; + const end = start + try self.readPackageLength(); try self.object(scope); // predicate self.termList(end, scope); } - fn opLnot(self: *Parser, scope: *Node) Error!void { + fn parseLnot(self: *Parser, scope: *Node) Error!void { // 0x92 followed by 0x93/94/95 is a compound comparison (two operands); // otherwise it is a plain LNot of one operand. const b = self.peek() orelse return error.Truncated; switch (b) { - op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => { - self.pos += 1; + opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => { + self.position += 1; try self.args(scope, 2); }, else => try self.object(scope), } } - fn opMatch(self: *Parser, scope: *Node) Error!void { + fn parseMatch(self: *Parser, scope: *Node) Error!void { try self.object(scope); // search package try self.skip(1); // match opcode 1 try self.object(scope); // operand 1 @@ -469,38 +469,38 @@ pub const Parser = struct { // --- extended opcodes (0x5B xx) ----------------------------------------- - fn opExt(self: *Parser, scope: *Node) Error!void { + fn parseExtended(self: *Parser, scope: *Node) Error!void { const e = try self.readByte(); switch (e) { - op.ext.mutex => try self.opMutex(scope), - op.ext.event => try self.opEvent(scope), - op.ext.op_region => try self.opRegion(scope), - op.ext.data_region => try self.opDataRegion(scope), - op.ext.field => try self.opField(scope, 1, false), - op.ext.index_field => try self.opField(scope, 2, false), - op.ext.bank_field => try self.opField(scope, 2, true), - op.ext.device => try self.opScopeLike(scope, .device), - op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone), - op.ext.processor => try self.opProcessor(scope), - op.ext.power_res => try self.opPowerRes(scope), + opcode.extended.mutex => try self.parseMutex(scope), + opcode.extended.event => try self.parseEvent(scope), + opcode.extended.operation_region => try self.parseRegion(scope), + opcode.extended.data_region => try self.parseDataRegion(scope), + opcode.extended.field => try self.parseField(scope, 1, false), + opcode.extended.index_field => try self.parseField(scope, 2, false), + opcode.extended.bank_field => try self.parseField(scope, 2, true), + opcode.extended.device => try self.parseScopeLike(scope, .device), + opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone), + opcode.extended.processor => try self.parseProcessor(scope), + opcode.extended.power_resource => try self.parsePowerResource(scope), - op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target - op.ext.create_field => try self.opCreateField(scope, 3), - op.ext.load_table => try self.args(scope, 6), - op.ext.load => try self.args(scope, 2), // NameString, Target - op.ext.stall, op.ext.sleep => try self.args(scope, 1), - op.ext.acquire => { + opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target + opcode.extended.create_field => try self.parseCreateField(scope, 3), + opcode.extended.load_table => try self.args(scope, 6), + opcode.extended.load => try self.args(scope, 2), // NameString, Target + opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1), + opcode.extended.acquire => { try self.object(scope); // mutex SuperName try self.skip(2); // timeout WordData }, - op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1), - op.ext.wait => try self.args(scope, 2), - op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2), - op.ext.fatal => { + opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1), + opcode.extended.wait => try self.args(scope, 2), + opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2), + opcode.extended.fatal => { try self.skip(5); // Type(byte) + Code(dword) try self.object(scope); // Arg TermArg }, - op.ext.revision, op.ext.debug, op.ext.timer => {}, + opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {}, else => return error.Malformed, } @@ -508,10 +508,10 @@ pub const Parser = struct { }; fn isNameStart(b: u8) bool { - return (b >= op.name_char_start and b <= op.name_char_end) or - b == op.name_char_underscore or - b == op.root_char or - b == op.parent_prefix_char or - b == op.dual_name_prefix or - b == op.multi_name_prefix; + return (b >= opcode.name_char_start and b <= opcode.name_char_end) or + b == opcode.name_char_underscore or + b == opcode.root_char or + b == opcode.parent_prefix_char or + b == opcode.dual_name_prefix or + b == opcode.multi_name_prefix; } diff --git a/src/device/device.zig b/src/device/device-model.zig similarity index 80% rename from src/device/device.zig rename to src/device/device-model.zig index 782d0ac..0676ee6 100644 --- a/src/device/device.zig +++ b/src/device/device-model.zig @@ -3,7 +3,7 @@ //! Discovery backends (ACPI today, device-tree later) translate their native //! hardware description into this one shape, so the rest of the kernel walks a //! plain `Device` tree without knowing which firmware described the machine — -//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `arch` +//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `architecture` //! applies to the CPU. //! //! This is deliberately minimal: enough to *describe* what was discovered (a @@ -14,14 +14,14 @@ const std = @import("std"); /// The hardware primitives a discovery backend needs but can't express portably. -/// The kernel injects an implementation (the arch VMM + port I/O), so the device -/// layer touches hardware without importing `arch` — the same discipline that lets +/// The kernel injects an implementation (the architecture VMM + port I/O), so the device +/// layer touches hardware without importing `architecture` — the same discipline that lets /// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4). pub const Hal = struct { /// Map a physical MMIO range and return the virtual address to reach it at. /// The device layer dereferences the returned address and never learns how /// the kernel places it (identity, physmap, a window — the kernel's choice). - mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64, + mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64, pioRead: *const fn (width: u8, port: u16) u32, pioWrite: *const fn (width: u8, port: u16, value: u32) void, }; @@ -74,29 +74,29 @@ pub const Ids = struct { pci_device: ?u16 = null, /// PCI class/subclass/prog-if packed as 0xCCSSPP. pci_class: ?u24 = null, - /// PCI bus/device/function packed as (bus << 8) | (dev << 3) | func — the key + /// PCI bus/device/function packed as (bus << 8) | (device << 3) | function — the key /// the ACPI address (`_ADR`) merge uses to match a namespace device to this node. pci_bdf: ?u16 = null, }; /// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is /// the busy case). Stored inline so a device is a single allocation. -pub const max_resources = 8; +pub const maximum_resources = 8; /// One node in the device tree. Nodes are individually heap-allocated and linked /// intrusively (first-child / next-sibling), the classic device-tree layout — /// no per-node dynamic arrays to manage. pub const Device = struct { - name_buf: [24]u8 = undefined, + name_buffer: [24]u8 = undefined, name_len: u8 = 0, class: DeviceClass = .unknown, ids: Ids = .{}, /// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like /// `name`; empty when unset. The generic layer stores/prints it without knowing /// how a backend encoded it. - hid_buf: [8]u8 = undefined, + hid_buffer: [8]u8 = undefined, hid_len: u8 = 0, - resources: [max_resources]Resource = undefined, + resources: [maximum_resources]Resource = undefined, resource_count: u8 = 0, parent: ?*Device = null, @@ -106,30 +106,30 @@ pub const Device = struct { /// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline /// buffer, so it stays valid for the life of the node with no extra allocation. pub fn name(self: *const Device) []const u8 { - return self.name_buf[0..self.name_len]; + return self.name_buffer[0..self.name_len]; } fn setName(self: *Device, s: []const u8) void { - const n: u8 = @intCast(@min(s.len, self.name_buf.len)); - @memcpy(self.name_buf[0..n], s[0..n]); + const n: u8 = @intCast(@min(s.len, self.name_buffer.len)); + @memcpy(self.name_buffer[0..n], s[0..n]); self.name_len = n; } /// The device's hardware id string, or empty if none is set. pub fn hid(self: *const Device) []const u8 { - return self.hid_buf[0..self.hid_len]; + return self.hid_buffer[0..self.hid_len]; } pub fn setHid(self: *Device, s: []const u8) void { - const n: u8 = @intCast(@min(s.len, self.hid_buf.len)); - @memcpy(self.hid_buf[0..n], s[0..n]); + const n: u8 = @intCast(@min(s.len, self.hid_buffer.len)); + @memcpy(self.hid_buffer[0..n], s[0..n]); self.hid_len = n; } - /// Record a resource. Silently drops beyond `max_resources` — discovery logs + /// Record a resource. Silently drops beyond `maximum_resources` — discovery logs /// the truncation rather than failing the whole tree. pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool { - if (self.resource_count >= max_resources) return false; + if (self.resource_count >= maximum_resources) return false; self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len }; self.resource_count += 1; return true; @@ -170,17 +170,17 @@ pub const DeviceTree = struct { self: *DeviceTree, parent: *Device, class: DeviceClass, - dev_name: []const u8, + device_name: []const u8, ) !*Device { const d = try self.allocator.create(Device); d.* = .{ .class = class, .parent = parent }; - d.setName(dev_name); + d.setName(device_name); if (parent.first_child == null) { parent.first_child = d; } else { - var cur = parent.first_child.?; - while (cur.next_sibling) |sib| cur = sib; - cur.next_sibling = d; + var current = parent.first_child.?; + while (current.next_sibling) |sib| current = sib; + current.next_sibling = d; } return d; } @@ -202,18 +202,18 @@ fn firstOfClassIn(node: *Device, class: DeviceClass) ?*Device { return null; } -fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void) void { +fn dumpNode(device: *const Device, depth: usize, emit: *const fn ([]const u8) void) void { const indent = @min(depth * 2, 40); - var buf: [200]u8 = undefined; - @memset(buf[0..indent], ' '); - const body = if (dev.hid_len != 0) - std.fmt.bufPrint(buf[indent..], "{s} [{s}] hid={s}\n", .{ dev.name(), @tagName(dev.class), dev.hid() }) catch return + var buffer: [200]u8 = undefined; + @memset(buffer[0..indent], ' '); + const body = if (device.hid_len != 0) + std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s}\n", .{ device.name(), @tagName(device.class), device.hid() }) catch return else - std.fmt.bufPrint(buf[indent..], "{s} [{s}]\n", .{ dev.name(), @tagName(dev.class) }) catch return; - emit(buf[0 .. indent + body.len]); + std.fmt.bufPrint(buffer[indent..], "{s} [{s}]\n", .{ device.name(), @tagName(device.class) }) catch return; + emit(buffer[0 .. indent + body.len]); - for (dev.resources[0..dev.resource_count]) |r| { + for (device.resources[0..device.resource_count]) |r| { var rbuf: [200]u8 = undefined; const pad = @min(indent + 2, 42); @memset(rbuf[0..pad], ' '); @@ -225,6 +225,6 @@ fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void) emit(rbuf[0 .. pad + rline.len]); } - var child = dev.first_child; + var child = device.first_child; while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit); } diff --git a/src/device/devicetree.zig b/src/device/device-tree.zig similarity index 73% rename from src/device/devicetree.zig rename to src/device/device-tree.zig index a1cfccb..154fa59 100644 --- a/src/device/devicetree.zig +++ b/src/device/device-tree.zig @@ -7,10 +7,10 @@ //! already routes to a backend rather than hard-coding ACPI — wiring the FDT //! parser in later is a local change here, not an architectural one. -const device = @import("device.zig"); +const device_model = @import("device-model.zig"); -/// Populate `dt` from a device-tree blob. Not implemented yet. -pub fn discover(dt: *device.DeviceTree) !void { - _ = dt; +/// Populate `device_tree` from a device-tree blob. Not implemented yet. +pub fn discover(device_tree: *device_model.DeviceTree) !void { + _ = device_tree; return error.Unsupported; } diff --git a/src/device/platform.zig b/src/device/platform.zig index 2f50ea0..c96c8b7 100644 --- a/src/device/platform.zig +++ b/src/device/platform.zig @@ -1,42 +1,42 @@ //! The firmware-agnostic discovery facade. //! //! The kernel calls `platform.discover()` and gets back a generic `DeviceTree` -//! without ever naming ACPI or device-tree — the same way it imports `arch` +//! without ever naming ACPI or device-tree — the same way it imports `architecture` //! without naming x86_64. Which backend runs is decided *at runtime* from what //! the bootloader handed us (an ACPI RSDP today, a device-tree blob later), //! because a single image — a future ARM kernel especially — may boot under -//! either firmware. That's a deliberate divergence from `arch`, which is a +//! either firmware. That's a deliberate divergence from `architecture`, which is a //! compile-time choice. const std = @import("std"); const danos = @import("danos"); -const device = @import("device.zig"); +const device_model = @import("device-model.zig"); const acpi = @import("acpi.zig"); const power = @import("power.zig"); -const devicetree = @import("devicetree.zig"); +const devicetree = @import("device-tree.zig"); -pub const DeviceTree = device.DeviceTree; -pub const Device = device.Device; -pub const DeviceClass = device.DeviceClass; -pub const Resource = device.Resource; -pub const ResourceKind = device.ResourceKind; -pub const Hal = device.Hal; -pub const PowerInfo = acpi.PowerInfo; +pub const DeviceTree = device_model.DeviceTree; +pub const Device = device_model.Device; +pub const DeviceClass = device_model.DeviceClass; +pub const Resource = device_model.Resource; +pub const ResourceKind = device_model.ResourceKind; +pub const Hal = device_model.Hal; +pub const PowerInformation = acpi.PowerInformation; pub const AmlStats = acpi.AmlStats; -pub const PlatformInfo = acpi.PlatformInfo; -pub const RegAccess = acpi.RegAccess; +pub const PlatformInformation = acpi.PlatformInformation; +pub const RegisterAccess = acpi.RegisterAccess; pub const IsoEntry = acpi.IsoEntry; pub const Cpu = acpi.Cpu; /// The register map + sleep types discovery extracted, for logging/diagnostics. -pub fn powerInfo() PowerInfo { - return acpi.power_info; +pub fn powerInformation() PowerInformation { + return acpi.power_information; } -/// The scalar firmware facts the arch layer needs to avoid legacy assumptions +/// The scalar firmware facts the architecture layer needs to avoid legacy assumptions /// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides). -pub fn platformInfo() PlatformInfo { - return acpi.platform_info; +pub fn platformInformation() PlatformInformation { + return acpi.platform_information; } /// AML parse integrity/diagnostics (namespace node count, bytes consumed). @@ -51,36 +51,36 @@ pub fn amlStats() AmlStats { /// bootstrap processor is actually running, so starting the rest is the pending SMP /// step (see docs/smp.md). Borrowed from static storage populated by `discover`. pub fn cpus() []const Cpu { - return acpi.cpu_info.cpus[0..acpi.cpu_info.count]; + return acpi.cpu_information.cpus[0..acpi.cpu_information.count]; } /// Non-zero only if enumeration found more processors than the static pool holds /// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent. pub fn cpusDropped() usize { - return acpi.cpu_info.dropped; + return acpi.cpu_information.dropped; } /// Enumerate hardware into a fresh device tree. `hal` supplies the hardware -/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for -/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up +/// primitives the backend needs (MMIO mapping for PCIe configuration space, port I/O for +/// ACPI registers); pass the architecture implementation. Errors leave nothing to clean up /// beyond the tree's own allocations. pub fn discover( - boot_info: *const danos.BootInfo, + boot_information: *const danos.BootInformation, allocator: std.mem.Allocator, hal: Hal, ) !DeviceTree { - var dt = try DeviceTree.init(allocator); + var device_tree = try DeviceTree.init(allocator); - if (boot_info.acpi_rsdp != 0) { - try acpi.discover(boot_info.acpi_rsdp, &dt, hal); + if (boot_information.acpi_rsdp != 0) { + try acpi.discover(boot_information.acpi_rsdp, &device_tree, hal); } else { // No ACPI RSDP. A device-tree boot would parse its blob here; today that // path is a stub, so this reports the machine described itself no way we // understand yet. - try devicetree.discover(&dt); + try devicetree.discover(&device_tree); } - return dt; + return device_tree; } /// Restart the machine. Never returns on success; returns only if no reset method diff --git a/src/device/power.zig b/src/device/power.zig index 4ad7796..8f8284b 100644 --- a/src/device/power.zig +++ b/src/device/power.zig @@ -9,8 +9,8 @@ //! re-initialisation, a milestone of its own. const acpi = @import("acpi.zig"); -const device = @import("device.zig"); -const Hal = device.Hal; +const device_model = @import("device-model.zig"); +const Hal = device_model.Hal; const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active @@ -19,27 +19,27 @@ const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is acti /// register is live. A no-op when the firmware exposes no SMI command port (ACPI /// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first. pub fn enable(hal: Hal) void { - const pi = acpi.power_info; + const pi = acpi.power_information; if (!pi.pm1a_cnt.present()) return; - if (readReg(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode + if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable); var spins: usize = 0; - while (readReg(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {} + while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {} } /// Restart the machine. Tries the ACPI reset register first, then the two legacy /// fallbacks. Returns only if every method failed (very unlikely). pub fn reboot(hal: Hal) void { - const pi = acpi.power_info; + const pi = acpi.power_information; // 1. The FADT reset register, when the firmware advertises support. if (pi.reset_supported and pi.reset.present()) { - writeReg(hal, pi.reset, pi.reset_value); + writeRegister(hal, pi.reset, pi.reset_value); delay(); } - // 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYS_RST). + // 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYSTEM_RST). hal.pioWrite(1, 0xCF9, 0x0E); hal.pioWrite(1, 0xCF9, 0x06); delay(); @@ -52,14 +52,14 @@ pub fn reboot(hal: Hal) void { /// it wasn't found in the AML, there is nothing safe to do and this returns. pub fn shutdown(hal: Hal) void { enable(hal); - const pi = acpi.power_info; + const pi = acpi.power_information; const s5 = pi.s5 orelse return; if (pi.pm1a_cnt.present()) { - writeReg(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a)); + writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a)); } if (pi.pm1b_cnt.present()) { - writeReg(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b)); + writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b)); } delay(); } @@ -76,25 +76,25 @@ fn sleepValue(slp_typ: u8) u32 { return (@as(u32, slp_typ & 0x7) << 10) | slp_en; } -fn readReg(hal: Hal, reg: acpi.RegAccess) u32 { - if (reg.mmio) { - const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true)); +fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 { + if (register.mmio) { + const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true)); return p.*; } - return hal.pioRead(reg.width, @intCast(reg.address)); + return hal.pioRead(register.width, @intCast(register.address)); } -fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void { - if (reg.mmio) { - const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true)); +fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void { + if (register.mmio) { + const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true)); p.* = value; } else { - hal.pioWrite(reg.width, @intCast(reg.address), value); + hal.pioWrite(register.width, @intCast(register.address), value); } } /// A short busy-wait so a reset/power-off takes effect before we fall through to -/// the next method. The empty asm is an arch-neutral barrier that keeps the loop +/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop /// from being optimised away. fn delay() void { var i: usize = 0; diff --git a/src/kernel/arch/x86_64/apic.zig b/src/kernel/arch/x86_64/apic.zig index 04621c7..82bae17 100644 --- a/src/kernel/arch/x86_64/apic.zig +++ b/src/kernel/arch/x86_64/apic.zig @@ -18,17 +18,17 @@ pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool }; // Platform facts from discovery (set by `configure` before bring-up). Defaults are // the legacy-safe assumptions so the code still works if discovery never ran. -var cfg_pic_present: bool = true; -var cfg_hpet_base: u64 = 0; // 0 = no HPET discovered -var cfg_pm_timer: ?PmTimer = null; +var configuration_pic_present: bool = true; +var configuration_hpet_base: u64 = 0; // 0 = no HPET discovered +var configuration_pm_timer: ?PmTimer = null; /// Which reference the last calibration used, for logging. var cal_source: []const u8 = "none"; /// Hand the LAPIC bring-up the discovered platform facts. Call before `init`. pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void { - cfg_pic_present = pic_present; - cfg_hpet_base = hpet_base; - cfg_pm_timer = pm_timer; + configuration_pic_present = pic_present; + configuration_hpet_base = hpet_base; + configuration_pm_timer = pm_timer; } /// The calibration reference the timer was measured against ("cpuid"/"hpet"/…). @@ -43,14 +43,15 @@ pub const timer_vector = 32; const spurious_vector = 47; // LAPIC register offsets. -const reg_spurious = 0x0F0; -const reg_eoi = 0x0B0; -const reg_icr_low = 0x300; // interrupt command register, low dword (writing it sends) -const reg_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31) -const reg_lvt_timer = 0x320; -const reg_timer_initial = 0x380; -const reg_timer_current = 0x390; -const reg_timer_divide = 0x3E0; +const register_spurious = 0x0F0; +const register_eoi = 0x0B0; +const register_id = 0x020; // this core's LAPIC id, in bits 24-31 +const register_icr_low = 0x300; // interrupt command register, low dword (writing it sends) +const register_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31) +const register_lvt_timer = 0x320; +const register_timer_initial = 0x380; +const register_timer_current = 0x390; +const register_timer_divide = 0x3E0; const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight @@ -90,11 +91,11 @@ fn rdtsc() u64 { return (@as(u64, high) << 32) | low; } -fn read(reg: u32) u32 { - return @as(*volatile u32, @ptrFromInt(base + reg)).*; +fn read(register: u32) u32 { + return @as(*volatile u32, @ptrFromInt(base + register)).*; } -fn write(reg: u32, value: u32) void { - @as(*volatile u32, @ptrFromInt(base + reg)).* = value; +fn write(register: u32, value: u32) void { + @as(*volatile u32, @ptrFromInt(base + register)).* = value; } /// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception @@ -116,14 +117,14 @@ fn remapAndMaskPic() void { /// UEFI Class 3 machine may have none), set the global-enable MSR bit, and /// software-enable the APIC via its spurious-vector register. pub fn init() void { - if (cfg_pic_present) remapAndMaskPic(); + if (configuration_pic_present) remapAndMaskPic(); const msr = io.rdmsr(ia32_apic_base_msr); // Reach the LAPIC through the physmap (paging.init maps its page there). - base = @intCast(danos.physToVirt(msr & 0xFFFFF000)); // physical base is bits 12+ + base = @intCast(danos.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+ io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable - write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable + write(register_spurious, 0x100 | spurious_vector); // bit 8 = software enable } /// Software-enable *this* core's Local APIC — the application-processor counterpart @@ -133,7 +134,7 @@ pub fn init() void { pub fn initSecondary() void { const msr = io.rdmsr(ia32_apic_base_msr); io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable - write(reg_spurious, 0x100 | spurious_vector); // software enable + write(register_spurious, 0x100 | spurious_vector); // software enable } // --- application-processor wakeup (INIT–SIPI–SIPI) -------------------------- @@ -141,8 +142,8 @@ pub fn initSecondary() void { /// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of /// the wake sequence. Blocks until the LAPIC reports the IPI was delivered. pub fn sendInit(apic_id: u32) void { - write(reg_icr_high, apic_id << 24); - write(reg_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered + write(register_icr_high, apic_id << 24); + write(register_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered waitIcrIdle(); } @@ -150,13 +151,13 @@ pub fn sendInit(apic_id: u32) void { /// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is /// sent twice after the INIT; both calls block until delivery completes. pub fn sendStartup(apic_id: u32, vector: u8) void { - write(reg_icr_high, apic_id << 24); - write(reg_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector + write(register_icr_high, apic_id << 24); + write(register_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector waitIcrIdle(); } fn waitIcrIdle() void { - while (read(reg_icr_low) & icr_delivery_pending != 0) {} + while (read(register_icr_low) & icr_delivery_pending != 0) {} } /// The calibration window: we time everything against a 10 ms reference interval. @@ -180,7 +181,7 @@ pub fn calibrate() void { } // 2. The discovered HPET. - if (!done and cfg_hpet_base != 0) { + if (!done and configuration_hpet_base != 0) { if (hpetHz()) |hpet_hz| { measure(hpet_hz, hpetMask(), readHpet); cal_source = "hpet"; @@ -190,7 +191,7 @@ pub fn calibrate() void { // 3. The ACPI PM timer (fixed 3.579545 MHz). if (!done) { - if (cfg_pm_timer) |pt| { + if (configuration_pm_timer) |pt| { measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer); cal_source = "pm-timer"; done = true; @@ -212,23 +213,23 @@ pub fn calibrate() void { tsc_base = rdtsc(); // the clock's zero point (boot) } -/// Run the LAPIC timer one-shot from its max count while a monotonic reference +/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference /// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and /// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`. fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void { const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms - write(reg_timer_divide, timer_divide_16); - write(reg_lvt_timer, lvt_masked); - write(reg_timer_initial, 0xFFFFFFFF); + write(register_timer_divide, timer_divide_16); + write(register_lvt_timer, lvt_masked); + write(register_timer_initial, 0xFFFFFFFF); const ref0 = refNow(); const tsc0 = rdtsc(); while (((refNow() -% ref0) & ref_mask) < calib_ticks) {} const tsc1 = rdtsc(); - const elapsed = 0xFFFFFFFF - read(reg_timer_current); - write(reg_timer_initial, 0); + const elapsed = 0xFFFFFFFF - read(register_timer_current); + write(register_timer_initial, 0); ticks_per_ms = elapsed / calib_ms; tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms); @@ -240,9 +241,9 @@ fn calibratePit() void { const pit_hz = 1_193_182; const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms); - write(reg_timer_divide, timer_divide_16); - write(reg_lvt_timer, lvt_masked); - write(reg_timer_initial, 0xFFFFFFFF); + write(register_timer_divide, timer_divide_16); + write(register_lvt_timer, lvt_masked); + write(register_timer_initial, 0xFFFFFFFF); io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0 @@ -255,8 +256,8 @@ fn calibratePit() void { while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded const tsc_end = rdtsc(); - const elapsed = 0xFFFFFFFF - read(reg_timer_current); - write(reg_timer_initial, 0); + const elapsed = 0xFFFFFFFF - read(register_timer_current); + write(register_timer_initial, 0); ticks_per_ms = elapsed / calib_ms; tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms); @@ -292,19 +293,19 @@ fn cpuid(leaf: u32) CpuidRegs { } // HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 = -// 64-bit-counter capable), general config at +0x10, main counter at +0xF0. +// 64-bit-counter capable), general configuration at +0x10, main counter at +0xF0. fn hpetRead64(off: usize) u64 { - return @as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).*; + return @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).*; } fn hpetWrite64(off: usize, value: u64) void { - @as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).* = value; + @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value; } /// Map + enable the HPET and return its tick frequency, or null if unusable. -/// Maps the HPET into the physmap and switches cfg_hpet_base to that virtual +/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual /// address, so the register accessors reach it without the identity map. fn hpetHz() ?u64 { - cfg_hpet_base = paging.mapMmio(cfg_hpet_base, 0x400, true); + configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true); const caps = hpetRead64(0x00); const period_fs = caps >> 32; // femtoseconds per tick if (period_fs == 0) return null; @@ -322,7 +323,7 @@ fn readHpet() u64 { } fn readPmTimer() u64 { - const pt = cfg_pm_timer.?; + const pt = configuration_pm_timer.?; // MMIO PM timer via the physmap (mapMmio is idempotent); the common case is // a legacy I/O port. if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*; @@ -334,9 +335,9 @@ fn readPmTimer() u64 { pub fn initTimer(hz: u32) void { timer_hz = hz; const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second - write(reg_timer_divide, timer_divide_16); - write(reg_lvt_timer, timer_vector | lvt_periodic); - write(reg_timer_initial, @intCast(count)); + write(register_timer_divide, timer_divide_16); + write(register_lvt_timer, timer_vector | lvt_periodic); + write(register_timer_initial, @intCast(count)); } /// Configured periodic-interrupt frequency (Hz). @@ -376,7 +377,7 @@ pub fn millis() u64 { /// Acknowledge the current interrupt so the LAPIC will deliver the next one. pub fn eoi() void { - write(reg_eoi, 0); + write(register_eoi, 0); } /// Optional callback run each tick (the scheduler registers it for preemption). @@ -390,10 +391,21 @@ pub fn setTickHook(hook: *const fn () void) void { /// tick hook (which may switch tasks). The interrupt is already acknowledged by /// the dispatcher before we get here, so a task switch here doesn't stall it. pub fn timerTick() void { + // Acknowledge before the tick hook: `on_tick` is the scheduler, which may switch + // tasks and not return promptly, and the LAPIC mustn't wait on it to deliver the + // next interrupt. (Each device handler now owns its own EOI — see + // `idt.interruptDispatch` — because a *routed* interrupt must be masked at the + // I/O APIC before it is acknowledged, an ordering the dispatcher can't impose.) + eoi(); tick_count +%= 1; if (on_tick) |hook| hook(); } +/// This core's Local APIC id — the interrupt destination for `routeGsi`. +pub fn localId() u8 { + return @truncate(read(register_id) >> 24); +} + /// Number of timer ticks so far. Volatile load: the count is bumped /// asynchronously by the interrupt handler, so callers must re-read memory. pub fn ticks() u64 { diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index 5c73945..7f3d167 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -1,11 +1,11 @@ -//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports -//! it as `@import("arch")` and never names x86_64 directly, so a second +//! x86_64 CPU operations. This is the "architecture" module: the generic kernel imports +//! it as `@import("architecture")` and never names x86_64 directly, so a second //! architecture is added by pointing that module at a different directory in //! build.zig — no change to the generic code. Keep everything CPU-specific here //! (halt, the descriptor tables, later paging), and nothing generic. const danos = @import("danos"); -const config = @import("config"); +const parameters = @import("parameters"); const gdt = @import("gdt.zig"); const tss = @import("tss.zig"); const idt = @import("idt.zig"); @@ -15,7 +15,7 @@ const apic = @import("apic.zig"); const ioapic = @import("ioapic.zig"); const io = @import("io.zig"); const smp = @import("smp.zig"); -const pcpu = @import("percpu.zig"); +const pcpu = @import("per-cpu.zig"); /// The saved register/trap frame passed to a fault handler. pub const CpuState = idt.CpuState; @@ -50,18 +50,18 @@ pub fn faultAddress(state: *const CpuState) ?u64 { ); } -// --- syscall ABI ------------------------------------------------------------ +// --- system_call ABI ------------------------------------------------------------ // The System V-style register convention (number in rax, arguments in // rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic // dispatcher never names a register. -/// The syscall number the user program passed. -pub fn syscallNumber(state: *const CpuState) u64 { +/// The system_call number the user program passed. +pub fn systemCallNumber(state: *const CpuState) u64 { return state.rax; } -/// Positional syscall argument `n`. -pub fn syscallArg(state: *const CpuState, n: u8) u64 { +/// Positional system_call argument `n`. +pub fn systemCallArg(state: *const CpuState, n: u8) u64 { return switch (n) { 0 => state.rdi, 1 => state.rsi, @@ -73,17 +73,17 @@ pub fn syscallArg(state: *const CpuState, n: u8) u64 { }; } -/// Write the syscall's return value into the frame — the entry paths restore +/// Write the system_call's return value into the frame — the entry paths restore /// user registers from it. -pub fn setSyscallResult(state: *CpuState, value: u64) void { +pub fn setSystemCallResult(state: *CpuState, value: u64) void { state.rax = value; } -/// Write a *second* syscall return value (rdx here — restored by both the -/// syscall/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by +/// Write a *second* system_call return value (rdx here — restored by both the +/// system_call/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by /// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the /// message length in rax. -pub fn setSyscallResult2(state: *CpuState, value: u64) void { +pub fn setSystemCallResult2(state: *CpuState, value: u64) void { state.rdx = value; } @@ -126,14 +126,14 @@ pub fn init() void { gdt.init(); tss.init(); idt.init(); - pcpu.initSyscall(); + pcpu.initSystemCall(); } /// Build the kernel's own page tables (with real permissions) and switch onto /// them. Needs the frame allocator and the boot info (for the memory map and the /// kernel's segment layout). Call once the frame allocator is up. -pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void { - paging.init(allocFrame, freeFrame, boot_info); +pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void { + paging.init(allocFrame, freeFrame, boot_information); } /// Create a new address space (returns the physical address of its root table — @@ -150,26 +150,26 @@ pub fn destroyAddressSpace(root: u64) void { } /// Map a user page into address space `root` (W^X is the caller's contract). -pub fn mapUserPageInto(root: u64, virt: u64, phys: u64, writable: bool, executable: bool) void { - paging.mapUserInto(root, virt, phys, writable, executable); +pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, executable: bool) void { + paging.mapUserInto(root, virtual, physical, writable, executable); } /// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX, /// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough. -pub fn mapUserDeviceInto(root: u64, virt: u64, phys: u64, len: u64) void { - paging.mapUserDeviceInto(root, virt, phys, len); +pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void { + paging.mapUserDeviceInto(root, virtual, physical, len); } /// Map a page into the kernel address space (non-executable). For the heap, etc. -pub fn mapPage(virt: u64, phys: u64, writable: bool) void { - paging.map(virt, phys, writable); +pub fn mapPage(virtual: u64, physical: u64, writable: bool) void { + paging.map(virtual, physical, writable); } /// Map a device MMIO range and return the virtual address to reach it at. This /// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and /// never exposes how the mapping is placed. -pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 { - return paging.mapMmio(phys, len, writable); +pub fn mapMmio(physical: u64, len: u64, writable: bool) u64 { + return paging.mapMmio(physical, len, writable); } /// The kernel's page-table root (physical), shared into every address space. @@ -192,7 +192,7 @@ pub fn activePageTable() u64 { /// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions: /// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the -/// per-CPU `kernel_rsp` (for the syscall stub, which switches by hand). Updated +/// per-CPU `kernel_rsp` (for the system_call stub, which switches by hand). Updated /// by the scheduler when it switches to a user task. pub fn setKernelStack(cpu: usize, top: usize) void { tss.rsp0Ptr(cpu).* = top; @@ -200,27 +200,27 @@ pub fn setKernelStack(cpu: usize, top: usize) void { } /// Remove a kernel mapping. -pub fn unmapPage(virt: u64) void { - paging.unmap(virt); +pub fn unmapPage(virtual: u64) void { + paging.unmap(virtual); } /// Remove a page mapping from address space `root` (for munmap of user pages). /// Clears the leaf entry only; freeing the underlying frame is the caller's job. -pub fn unmapUserPageInto(root: u64, virt: u64) void { - paging.unmapInto(root, virt); +pub fn unmapUserPageInto(root: u64, virtual: u64) void { + paging.unmapInto(root, virtual); } -/// Resolve `virt` to its physical address in the address space rooted at `root` +/// Resolve `virtual` to its physical address in the address space rooted at `root` /// (any address space, not just the live one), or null if unmapped. Used to find /// the frame behind a user page for munmap, and for cross-address-space copies. -pub fn translate(root: u64, virt: u64) ?u64 { - return paging.translateIn(root, virt); +pub fn translate(root: u64, virtual: u64) ?u64 { + return paging.translateIn(root, virtual); } /// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps /// W^X: code read-only + executable, data writable + no-execute. -pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void { - paging.mapUser(virt, phys, writable, executable); +pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void { + paging.mapUser(virtual, physical, writable, executable); } // --- ring 3 entry/exit ----------------------------------------------------- @@ -233,27 +233,27 @@ pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void; /// Abandon the in-flight ring-3 trap context and resume the kernel as if -/// `enter_user` had returned (defined in isr.s). Called by the exit syscall. +/// `enter_user` had returned (defined in isr.s). Called by the exit system_call. extern fn user_exit_to_kernel() callconv(.c) noreturn; /// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the -/// caller's CPU index; the arch layer can't ask the scheduler). Returns after the -/// user program exits via syscall. Interrupts are disabled on return (the exit +/// caller's CPU index; the architecture layer can't ask the scheduler). Returns after the +/// user program exits via system_call. Interrupts are disabled on return (the exit /// arrives through an interrupt gate) — the caller re-enables. pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void { enter_user(entry, stack_top, tss.rsp0Ptr(cpu)); } /// Never returns to the user program: unwind to the kernel context that called -/// `enterUser`. For the exit syscall's handler. +/// `enterUser`. For the exit system_call's handler. pub fn userExit() noreturn { user_exit_to_kernel(); } -/// Register the handler for the user syscall gate (int 0x80, vector 128). The -/// handler may write the trap frame (see `setSyscallResult`). -pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void { - idt.setSyscallHandler(handler); +/// Register the handler for the user system_call gate (int 0x80, vector 128). The +/// handler may write the trap frame (see `setSystemCallResult`). +pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void { + idt.setSystemCallHandler(handler); } /// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each @@ -266,7 +266,7 @@ pub fn setCpuLocal(cpu: usize, ptr: usize) void { /// This core's scheduler pointer (via the GS base) — a per-core register, so each /// core sees its own without locking. Valid in any ring-0 context. pub fn cpuLocal() usize { - return pcpu.sched(); + return pcpu.scheduler(); } // --- SMP: application-processor bring-up ---------------------------------- @@ -275,8 +275,8 @@ pub fn cpuLocal() usize { /// The frame stays inert (zeroed, non-executable) between wakes and is armed only /// while a core is climbing — so a core can be (re)woken at any time (retry, or a /// future power manager) without leaving an executable page resident. See smp.zig. -pub fn setTrampolinePage(phys: u64) void { - smp.setTrampolinePage(phys); +pub fn setTrampolinePage(physical: u64) void { + smp.setTrampolinePage(physical); } /// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on @@ -290,7 +290,7 @@ pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize) return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4()); } -/// Register the generic entry a woken AP jumps to once its arch state is up (its own +/// Register the generic entry a woken AP jumps to once its architecture state is up (its own /// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here. pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void { smp.setSecondaryEntry(entry); @@ -316,22 +316,22 @@ pub fn trampolinePage() u64 { return smp.trampolinePage(); } -/// Whether the page at `virt` is currently mapped executable (present, NX clear). -pub fn pageExecutable(virt: u64) bool { - return paging.isExecutable(virt); +/// Whether the page at `virtual` is currently mapped executable (present, NX clear). +pub fn pageExecutable(virtual: u64) bool { + return paging.isExecutable(virtual); } -/// Kernel tick rate (the scheduler's time quantum), from config. -pub const timer_hz = config.timer_hz; +/// Kernel tick rate (the scheduler's time quantum), from configuration. +pub const timer_hz = parameters.timer_hz; -/// The ACPI PM timer, as a calibration reference (re-exported for the config). +/// The ACPI PM timer, as a calibration reference (re-exported for the configuration). pub const PmTimer = apic.PmTimer; -/// A MADT interrupt-source override (re-exported for the config). +/// A MADT interrupt-source override (re-exported for the configuration). pub const IsoEntry = ioapic.IsoEntry; -/// Discovered platform facts the arch layer needs so it makes no legacy +/// Discovered platform facts the architecture layer needs so it makes no legacy /// assumptions — sourced from the device tree + ACPI, passed in by the kernel. -pub const PlatformConfig = struct { +pub const PlatformConfiguration = struct { /// Whether the legacy 8259 PIC is present (skip programming it if not). pic_present: bool = true, /// HPET MMIO base (0 = none) — a calibration reference for the timer. @@ -345,18 +345,18 @@ pub const PlatformConfig = struct { overrides: []const IsoEntry = &.{}, }; -/// Apply the discovered platform config. Must run before `startTimer` (the timer +/// Apply the discovered platform configuration. Must run before `startTimer` (the timer /// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing. /// Maps + masks the I/O APIC immediately. -pub fn configurePlatform(cfg: PlatformConfig) void { - apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer); - ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides); +pub fn configurePlatform(configuration: PlatformConfiguration) void { + apic.configure(configuration.pic_present, configuration.hpet_base, configuration.pm_timer); + ioapic.configure(configuration.ioapic_base, configuration.ioapic_gsi_base, configuration.overrides); ioapic.init(); } /// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port). -pub fn serialReconfigure(is_mmio: bool, addr: u64) void { - serial.reconfigure(is_mmio, addr); +pub fn serialReconfigure(is_mmio: bool, address: u64) void { + serial.reconfigure(is_mmio, address); } /// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…). @@ -373,6 +373,43 @@ pub fn irqRouteRaw(n: u32) u32 { return ioapic.entryLow(n); } +// --- device-IRQ plumbing, for src/kernel/irq.zig ----------------------------- +// +// The generic IRQ layer speaks GSIs and vectors; everything below hides the fact +// that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The +// vector window is bounded by the stubs isr.s actually emits: `gate_count` = 48, +// vector 32 is the LAPIC timer and 47 is the spurious vector, leaving 33..46. + +pub const irq_vector_base: u8 = 33; +pub const irq_vector_count: u8 = 14; // 33..46 inclusive + +/// True if `gsi` is one this machine's interrupt router can deliver. +pub fn irqOwnsGsi(gsi: u32) bool { + return ioapic.ownsGsi(gsi); +} + +/// Install `handler` on `vector` (an absolute IDT gate index). +pub fn irqSetHandler(vector: u8, handler: *const fn () void) void { + idt.setHandler(vector, handler); +} + +/// Route `gsi` to `vector` on *this* core, masked. Unmask with `irqUnmask` once bound. +pub fn irqRoute(gsi: u32, vector: u8, level: bool, active_low: bool) void { + ioapic.routeGsi(gsi, vector, apic.localId(), level, active_low); +} + +pub fn irqMask(gsi: u32) void { + ioapic.maskGsi(gsi); +} +pub fn irqUnmask(gsi: u32) void { + ioapic.unmaskGsi(gsi); +} + +/// Acknowledge the interrupt currently in service on this core's LAPIC. +pub fn irqEoi() void { + apic.eoi(); +} + /// Enable the Local APIC, calibrate its timer against the best available reference /// (see apic.calibrate — no longer the PIT by default), and start it firing at /// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be diff --git a/src/kernel/arch/x86_64/gdt.zig b/src/kernel/arch/x86_64/gdt.zig index d759370..2f57d68 100644 --- a/src/kernel/arch/x86_64/gdt.zig +++ b/src/kernel/arch/x86_64/gdt.zig @@ -9,7 +9,7 @@ //! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its //! own copy of the table with its own TSS descriptor. Slot 0 is the BSP. -const config = @import("config"); +const parameters = @import("parameters"); /// Selectors into the table (index * 8). Same on every core's GDT. pub const kernel_code = 0x08; @@ -24,7 +24,7 @@ pub const tss_selector = 0x28; pub const user_code_rpl3 = user_code | 3; pub const user_data_rpl3 = user_data | 3; -const max_cpus = config.max_cpus; +const maximum_cpus = parameters.maximum_cpus; const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high /// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor, @@ -47,7 +47,7 @@ const template = [entries]u64{ }; /// One GDT per core (each a copy of the template, differing only in its TSS slot). -var gdts = [_][entries]u64{template} ** max_cpus; +var gdts = [_][entries]u64{template} ** maximum_cpus; /// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task /// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit diff --git a/src/kernel/arch/x86_64/idt.zig b/src/kernel/arch/x86_64/idt.zig index f9fff47..43e60b6 100644 --- a/src/kernel/arch/x86_64/idt.zig +++ b/src/kernel/arch/x86_64/idt.zig @@ -9,7 +9,6 @@ const gdt = @import("gdt.zig"); const tss = @import("tss.zig"); -const apic = @import("apic.zig"); /// Highest vector we install a gate/stub for (exceptions 0-31 plus the device /// range 32-47, which covers the timer and the spurious vector). @@ -26,17 +25,17 @@ pub fn setHandler(vector: usize, handler: Handler) void { handlers[vector] = handler; } -/// The ring-3 syscall gate's vector (`int $0x80`, the classic choice — well away -/// from the device range) and its handler. Unlike device handlers, a syscall +/// The ring-3 system_call gate's vector (`int $0x80`, the classic choice — well away +/// from the device range) and its handler. Unlike device handlers, a system_call /// handler gets the (mutable) trap frame: it reads its arguments from the saved /// user registers and writes rax as the return value, which isr_common then /// restores into the user context. -pub const syscall_vector = 128; +pub const system_call_vector = 128; -var syscall_handler: ?*const fn (*CpuState) void = null; +var system_call_handler: ?*const fn (*CpuState) void = null; -pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void { - syscall_handler = handler; +pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void { + system_call_handler = handler; } /// The register + trap frame the ISR stubs build on the stack, laid out so the @@ -141,13 +140,13 @@ pub fn init() void { // Run the double-fault handler (vector 8) on IST1: a #DF usually means the // current stack is unusable, so it needs a guaranteed-good one. See tss.zig. idt[8].ist = tss.double_fault_ist; - // The syscall gate. Installed outside the 0..gate_count loop (stubs 48-127 + // The system_call gate. Installed outside the 0..gate_count loop (stubs 48-127 // don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a // #GP. An interrupt gate (not trap): IF is cleared for the handler, which // the ring-3 exit path relies on. - const syscall_stub = @extern(*const anyopaque, .{ .name = "isr128" }); - setGate(syscall_vector, @intFromPtr(syscall_stub)); - idt[syscall_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate + const system_call_stub = @extern(*const anyopaque, .{ .name = "isr128" }); + setGate(system_call_vector, @intFromPtr(system_call_stub)); + idt[system_call_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate loadOnThisCpu(); } @@ -168,15 +167,17 @@ pub fn loadOnThisCpu() void { export fn interruptDispatch(state: *CpuState) callconv(.c) void { if (state.vector < 32) { on_fault(state); // CPU exception — never returns - } else if (state.vector == syscall_vector) { + } else if (state.vector == system_call_vector) { // Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI. - if (syscall_handler) |handler| handler(state); + if (system_call_handler) |handler| handler(state); } else if (handlers[state.vector]) |handler| { - // Acknowledge before running the handler: a handler that switches tasks - // (the scheduler) may not return promptly, and the LAPIC mustn't wait on - // it to deliver the next interrupt. Fine for edge-triggered sources like - // the timer; a level-triggered device would need EOI after handling. - apic.eoi(); + // The handler owns its EOI. It used to be issued here, before the call — + // correct for the LAPIC timer, but impossible to reconcile with a + // level-triggered device line, which must be **masked at the I/O APIC + // before** it is acknowledged or it redelivers instantly and storms + // (the driver that would quiet it lives in ring 3 and hasn't run yet). + // Only the handler knows which discipline its source needs, so only the + // handler can sequence it. See apic.timerTick and irq.dispatch. handler(); } // else: spurious/unhandled device interrupt — don't acknowledge it diff --git a/src/kernel/arch/x86_64/ioapic.zig b/src/kernel/arch/x86_64/ioapic.zig index debf2cc..8c991c7 100644 --- a/src/kernel/arch/x86_64/ioapic.zig +++ b/src/kernel/arch/x86_64/ioapic.zig @@ -2,11 +2,15 @@ //! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from //! ACPI's MADT (via discovery), never assumed. //! -//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own -//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now -//! is `init`, which maps the I/O APIC and **masks every input**, the correct -//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real -//! device driver (a keyboard, say). +//! `init` maps the I/O APIC and **masks every input** — the correct quiescent state +//! on a legacy-free machine. Lines are then unmasked one at a time, as user-space +//! drivers bind them (`routeGsi`/`unmaskGsi`, driven by src/kernel/irq.zig). +//! +//! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ** +//! and resolves it through the MADT overrides — for in-kernel use, and still without +//! a caller. `routeGsi` takes a **GSI** directly, which is what a device's own +//! routing capability names (e.g. the HPET's `Tn_INT_ROUTE_CAP`), and is the path a +//! bound driver interrupt takes. const paging = @import("paging.zig"); @@ -16,14 +20,14 @@ pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 }; var base: u64 = 0; // 0 = no I/O APIC discovered var gsi_base: u32 = 0; -var max_entries: u32 = 0; +var maximum_entries: u32 = 0; var overrides: [16]IsoEntry = undefined; var override_count: usize = 0; // The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN). -const reg_ioregsel = 0x00; -const reg_iowin = 0x10; -const reg_version = 0x01; +const register_ioregsel = 0x00; +const register_iowin = 0x10; +const register_version = 0x01; const redir_base = 0x10; // redirection table: two 32-bit regs per entry const redir_mask = 1 << 16; // mask bit in the low dword @@ -35,18 +39,18 @@ pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso; } -fn regRead(index: u32) u32 { - @as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index; - return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*; +fn registerRead(index: u32) u32 { + @as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index; + return @as(*volatile u32, @ptrFromInt(base + register_iowin)).*; } -fn regWrite(index: u32, value: u32) void { - @as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index; - @as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value; +fn registerWrite(index: u32, value: u32) void { + @as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index; + @as(*volatile u32, @ptrFromInt(base + register_iowin)).* = value; } fn writeEntry(n: u32, low: u32, high: u32) void { - regWrite(redir_base + 2 * n, low); - regWrite(redir_base + 2 * n + 1, high); + registerWrite(redir_base + 2 * n, low); + registerWrite(redir_base + 2 * n + 1, high); } /// Map the I/O APIC and mask every redirection entry — the safe quiescent state. @@ -55,9 +59,9 @@ pub fn init() void { // Reach the I/O APIC through the physmap; switch `base` to that virtual // address so the register accessors work without the identity map. base = paging.mapMmio(base, 0x1000, true); - max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1; + maximum_entries = ((registerRead(register_version) >> 16) & 0xFF) + 1; var n: u32 = 0; - while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0); + while (n < maximum_entries) : (n += 1) writeEntry(n, redir_mask, 0); } /// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override @@ -76,7 +80,7 @@ pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void { } if (gsi < gsi_base) return; const n = gsi - gsi_base; - if (n >= max_entries) return; + if (n >= maximum_entries) return; // Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked. // MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level). @@ -87,13 +91,62 @@ pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void { writeEntry(n, low, high); } +// --- GSI-level control (the user-space driver path) -------------------------- +// +// `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides. +// A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the +// HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection +// entry. These three are what `src/kernel/irq.zig` drives. +// +// Callers must serialise: the I/O APIC is reached through an index/data register +// pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel +// holds the big lock across these. + +/// Redirection-entry index for `gsi`, or null if this I/O APIC doesn't own it. +fn entryFor(gsi: u32) ?u32 { + if (base == 0 or gsi < gsi_base) return null; + const n = gsi - gsi_base; + return if (n < maximum_entries) n else null; +} + +/// True if `gsi` lands on this I/O APIC — the kernel's validity check before binding. +pub fn ownsGsi(gsi: u32) bool { + return entryFor(gsi) != null; +} + +/// Point `gsi` at `vector` on the LAPIC `apic_id`, with explicit polarity/trigger, +/// and leave it **masked**. The caller unmasks once a handler is bound — otherwise a +/// device asserting between route and bind would fire into a null handler. +pub fn routeGsi(gsi: u32, vector: u8, apic_id: u8, level: bool, active_low: bool) void { + const n = entryFor(gsi) orelse return; + var low: u32 = @as(u32, vector) | redir_mask; // masked until bound + if (active_low) low |= (1 << 13); + if (level) low |= (1 << 15); + writeEntry(n, low, @as(u32, apic_id) << 24); +} + +/// Stop `gsi` reaching any CPU. Called from the ISR *before* the LAPIC EOI: a +/// level-triggered line is still asserted at that point, so an unmasked entry would +/// redeliver immediately and storm before the user-space driver ever runs. +pub fn maskGsi(gsi: u32) void { + const n = entryFor(gsi) orelse return; + registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) | redir_mask); +} + +/// Let `gsi` through again — the tail of `irq_ack`, once the driver has quieted the +/// device (so the line is deasserted and this can't immediately refire). +pub fn unmaskGsi(gsi: u32) void { + const n = entryFor(gsi) orelse return; + registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) & ~@as(u32, redir_mask)); +} + /// Number of redirection entries the I/O APIC advertises (0 until `init`). pub fn entryCount() u32 { - return max_entries; + return maximum_entries; } /// The low dword of redirection entry `n` — for diagnostics/read-back. pub fn entryLow(n: u32) u32 { if (base == 0) return 0; - return regRead(redir_base + 2 * n); + return registerRead(redir_base + 2 * n); } diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index c5113ae..f7f5093 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -23,7 +23,7 @@ const pwt: u64 = 1 << 3; // page write-through const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT) const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim const no_execute: u64 = 1 << 63; -const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; +const address_mask: u64 = 0x000F_FFFF_FFFF_F000; // ELF segment flags (p_flags). const pf_x: u32 = 1; @@ -54,11 +54,11 @@ const bootstrap_physmap_limit: u64 = 4 << 30; /// Dereference a page-table frame by its physical address, via the physmap. /// This is the single hinge for the higher-half move: page tables hold physical /// frame addresses (pmm gives out physical frames, and CR3/PTEs must be -/// physical), but the kernel reaches them at `physmap_base + phys`. Valid under +/// physical), but the kernel reaches them at `physmap_base + physical`. Valid under /// both the loader's bootstrap tables and the kernel's own, which share the /// physmap base. -fn tableAt(phys: u64) *[512]u64 { - return @ptrFromInt(danos.physToVirt(phys)); +fn tableAt(physical: u64) *[512]u64 { + return @ptrFromInt(danos.physicalToVirtual(physical)); } fn allocTable() u64 { @@ -73,41 +73,41 @@ fn allocTable() u64 { /// entries are writable and executable so the leaf's bits govern (a page is /// writable only if every level is; non-executable if any level is). fn descend(entry: *u64) u64 { - if (entry.* & present != 0) return entry.* & addr_mask; + if (entry.* & present != 0) return entry.* & address_mask; const frame = allocTable(); entry.* = frame | present | writable; return frame; } -/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added). -fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void { - const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF]; +/// Map one 4 KiB page `virtual` -> `physical` with `flags` (present is added). +fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void { + const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF]; // The kernel half is fixed after init: every top-half PML4 entry is // pre-created so address spaces can share it by copying these slots. A new // one here would be invisible to address spaces already made. - if (init_done and (virt >> 63) == 1 and pml4e.* & present == 0) + if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0) @panic("paging: new higher-half PML4 entry after init"); const pdpt = descend(pml4e); - const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF]; + const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF]; const pd = descend(pdpte); - const pde = &tableAt(pd)[(virt >> 21) & 0x1FF]; + const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF]; const pt = descend(pde); - tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present; + tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present; } -/// Map [phys_base, phys_base+len) into the physmap (at physToVirt(phys)) with +/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with /// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent /// window onto physical memory once the low identity map goes away. -fn mapRangePhysmap(pml4: u64, phys_base: u64, len: u64, flags: u64) void { - var addr = phys_base & ~@as(u64, page_size - 1); - const end = phys_base + len; - while (addr < end) : (addr += page_size) { - mapPage(pml4, danos.physToVirt(addr), addr, flags); +fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void { + var address = physical_base & ~@as(u64, page_size - 1); + const end = physical_base + len; + while (address < end) : (address += page_size) { + mapPage(pml4, danos.physicalToVirtual(address), address, flags); } } fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion { - return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len]; + return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len]; } /// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we @@ -118,36 +118,36 @@ fn enableNx() void { } /// Build the address space and switch onto it. -pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void { +pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void { alloc_frame = allocFrame; free_frame = freeFrame; enableNx(); const pml4 = allocTable(); - // 1. All RAM in the physmap (physToVirt(phys)) RW + NX. No identity/low-half + // 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half // mapping: the low half belongs to user space. MMIO is skipped here and // mapped on demand (mapMmio) or explicitly below. - for (regions(boot_info.memory_map)) |r| { + for (regions(boot_information.memory_map)) |r| { if (r.kind == .mmio) continue; mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute); } // 2. Physmap windows for the framebuffer and the Local APIC (device memory // the kernel touches directly), RW + NX. - const fb = boot_info.framebuffer; + const fb = boot_information.framebuffer; mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute); - mapPage(pml4, danos.physToVirt(0xFEE00000), 0xFEE00000, present | writable | no_execute); + mapPage(pml4, danos.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute); // 3. The kernel's own segments at their higher-half link addresses, mapped // to their low physical load addresses with real ELF permissions: code // R+X, rodata R, data R+W+NX. This is the W^X guarantee. - for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| { + for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| { var flags: u64 = present; if (seg.flags & pf_w != 0) flags |= writable; if (seg.flags & pf_x == 0) flags |= no_execute; var off: u64 = 0; while (off < seg.pages * page_size) : (off += page_size) { - mapPage(pml4, seg.virt + off, seg.phys + off, flags); + mapPage(pml4, seg.virtual + off, seg.physical + off, flags); } } @@ -187,30 +187,30 @@ pub fn loadCr3(pml4: u64) void { /// Map a page into the kernel address space on demand (for the heap, etc.). /// `writable_page` controls W; pages are always mapped non-executable. -pub fn map(virt: u64, phys: u64, writable_page: bool) void { +pub fn map(virtual: u64, physical: u64, writable_page: bool) void { var flags: u64 = present | no_execute; if (writable_page) flags |= writable; - mapPage(kernel_pml4, virt, phys, flags); - invalidate(virt); + mapPage(kernel_pml4, virtual, physical, flags); + invalidate(virtual); } /// Map a device MMIO range into the physmap and return the virtual address to -/// use for it (physToVirt(phys)). The single way the kernel (and the device +/// use for it (physicalToVirtual(physical)). The single way the kernel (and the device /// layer, via the HAL) reaches memory-mapped registers once the identity map is /// gone: physmap pages are RW + NX, so a driver never executes device memory. /// Idempotent for already-mapped ranges. `len` 0 maps one page. -pub fn mapMmio(phys: u64, len: u64, writable_page: bool) u64 { +pub fn mapMmio(physical: u64, len: u64, writable_page: bool) u64 { var flags: u64 = present | no_execute; if (writable_page) flags |= writable; - const first = phys & ~@as(u64, page_size - 1); - const last = phys + (if (len == 0) 1 else len) - 1; - var addr = first; - while (addr <= (last & ~@as(u64, page_size - 1))) : (addr += page_size) { - const virt = danos.physToVirt(addr); - mapPage(kernel_pml4, virt, addr, flags); - invalidate(virt); + const first = physical & ~@as(u64, page_size - 1); + const last = physical + (if (len == 0) 1 else len) - 1; + var address = first; + while (address <= (last & ~@as(u64, page_size - 1))) : (address += page_size) { + const virtual = danos.physicalToVirtual(address); + mapPage(kernel_pml4, virtual, address, flags); + invalidate(virtual); } - return danos.physToVirt(phys); + return danos.physicalToVirtual(physical); } /// Like `descend`, but also sets the U/S bit on the intermediate entry (new or @@ -223,51 +223,51 @@ fn descendUser(entry: *u64) u64 { return table; } -/// Map one 4 KiB page `virt` -> `phys` accessible from ring 3. W^X is the +/// Map one 4 KiB page `virtual` -> `physical` accessible from ring 3. W^X is the /// caller's contract: code pages are read-only + executable, data pages are -/// writable + no-execute. `virt` must lie in a user-exclusive region (see +/// writable + no-execute. `virtual` must lie in a user-exclusive region (see /// `descendUser`). -pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void { - mapUserInto(kernel_pml4, virt, phys, writable_page, executable); +pub fn mapUser(virtual: u64, physical: u64, writable_page: bool, executable: bool) void { + mapUserInto(kernel_pml4, virtual, physical, writable_page, executable); } /// Map a ring-3-accessible page into the address space rooted at `pml4` (which /// may be a process's own table or the kernel's). W^X is the caller's contract. -pub fn mapUserInto(pml4: u64, virt: u64, phys: u64, writable_page: bool, executable: bool) void { +pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool, executable: bool) void { var flags: u64 = present | user; if (writable_page) flags |= writable; if (!executable) flags |= no_execute; - const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF]; + const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF]; const pdpt = descendUser(pml4e); - const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF]; + const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF]; const pd = descendUser(pdpte); - const pde = &tableAt(pd)[(virt >> 21) & 0x1FF]; + const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF]; const pt = descendUser(pde); - tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags; - invalidate(virt); + tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags; + invalidate(virtual); } -/// Map a device MMIO window `[phys, phys+len)` into the user (low) half of the +/// Map a device MMIO window `[physical, physical+len)` into the user (low) half of the /// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages /// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and /// carry the `device_grant` bit so teardown does not return the MMIO frames to the -/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virt` in a -/// user-exclusive range (PML4[225]). Both `virt` and `phys` are page-aligned by -/// the caller; a sub-page `phys` offset is the caller's to re-apply. -pub fn mapUserDeviceInto(pml4: u64, virt: u64, phys: u64, len: u64) void { +/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a +/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by +/// the caller; a sub-page `physical` offset is the caller's to re-apply. +pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void { const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant; - const first = phys & ~@as(u64, page_size - 1); - const last = (phys + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1); + const first = physical & ~@as(u64, page_size - 1); + const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1); var off: u64 = 0; while (first + off <= last) : (off += page_size) { - const v = virt + off; + const v = virtual + off; const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF]; const pdpt = descendUser(pml4e); const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF]; const pd = descendUser(pdpte); const pde = &tableAt(pd)[(v >> 21) & 0x1FF]; const pt = descendUser(pde); - tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & addr_mask) | flags; + tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags; invalidate(v); } } @@ -291,39 +291,39 @@ pub fn createAddressSpace() ?u64 { pub fn destroyAddressSpace(pml4: u64) void { const t = tableAt(pml4); for (0..256) |i| { - if (t[i] & present != 0) freeSubtree(t[i] & addr_mask, 3); // PDPT level + if (t[i] & present != 0) freeSubtree(t[i] & address_mask, 3); // PDPT level } free_frame(pml4); } /// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At /// level 1 the entries are leaf data frames; above, they are child tables. -fn freeSubtree(phys: u64, level: u32) void { - const t = tableAt(phys); +fn freeSubtree(physical: u64, level: u32) void { + const t = tableAt(physical); for (t) |e| { if (e & present == 0) continue; if (level > 1) { - freeSubtree(e & addr_mask, level - 1); + freeSubtree(e & address_mask, level - 1); } else if (e & device_grant == 0) { // A device-grant leaf points at MMIO, not RAM — returning it to the // frame allocator would corrupt the pool. Only reclaim real RAM. - free_frame(e & addr_mask); + free_frame(e & address_mask); } } - free_frame(phys); // page-table frames are always real RAM + free_frame(physical); // page-table frames are always real RAM } -/// Whether `virt` is currently mapped **executable** — present with the NX bit +/// Whether `virtual` is currently mapped **executable** — present with the NX bit /// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page /// case). Returns false if unmapped. Used for W^X checks in tests. -pub fn isExecutable(virt: u64) bool { - const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF]; +pub fn isExecutable(virtual: u64) bool { + const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF]; if (pml4e & present == 0) return false; - const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF]; + const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF]; if (pdpte & present == 0) return false; - const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF]; + const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF]; if (pde & present == 0) return false; - const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF]; + const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF]; if (pte & present == 0) return false; return pte & no_execute == 0; } @@ -333,57 +333,57 @@ pub fn isExecutable(virt: u64) bool { /// application processors fetch the AP trampoline from a low RAM page under paging — /// so that one page must be executable. A deliberate, temporary W^X exception for a /// single bring-up page; the caller frees it once every AP is up. -pub fn setExecutable(phys: u64) void { - mapPage(kernel_pml4, phys, phys, present | writable); // note: no no_execute - invalidate(phys); +pub fn setExecutable(physical: u64) void { + mapPage(kernel_pml4, physical, physical, present | writable); // note: no no_execute + invalidate(physical); } /// Remove a mapping and flush it from the TLB. -pub fn unmap(virt: u64) void { - unmapInto(kernel_pml4, virt); +pub fn unmap(virtual: u64) void { + unmapInto(kernel_pml4, virtual); } /// Remove a mapping from the address space rooted at `pml4` (a process's own /// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE — /// the intermediate tables and any frame the PTE pointed at are left to the /// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables). -pub fn unmapInto(pml4: u64, virt: u64) void { - const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF]; +pub fn unmapInto(pml4: u64, virtual: u64) void { + const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF]; if (pml4e & present == 0) return; - const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF]; + const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF]; if (pdpte & present == 0) return; - const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF]; + const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF]; if (pde & present == 0) return; - tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0; - invalidate(virt); + tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF] = 0; + invalidate(virtual); } /// Resolve a virtual address to a physical one in the address space rooted at /// `pml4`, walking the tables through the physmap (CR3-independent — works for -/// any address space, not just the live one). Returns null if `virt` is not +/// any address space, not just the live one). Returns null if `virtual` is not /// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page /// case. The foundation for cross-address-space copies and for munmap (which /// needs the frame behind a user vaddr to free it). -pub fn translateIn(pml4: u64, virt: u64) ?u64 { - const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF]; +pub fn translateIn(pml4: u64, virtual: u64) ?u64 { + const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF]; if (pml4e & present == 0) return null; - const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF]; + const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF]; if (pdpte & present == 0) return null; - const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF]; + const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF]; if (pde & present == 0) return null; - const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF]; + const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF]; if (pte & present == 0) return null; - return (pte & addr_mask) | (virt & (page_size - 1)); + return (pte & address_mask) | (virtual & (page_size - 1)); } -fn invalidate(virt: u64) void { +fn invalidate(virtual: u64) void { // invlpg needs its operand via a register-indirect memory reference that Zig // inline asm won't form directly, so stage the address in a register first. asm volatile ( \\mov %[v], %%rax \\invlpg (%%rax) : - : [v] "r" (virt), + : [v] "r" (virtual), : .{ .rax = true, .memory = true } ); } diff --git a/src/kernel/arch/x86_64/percpu.zig b/src/kernel/arch/x86_64/per-cpu.zig similarity index 59% rename from src/kernel/arch/x86_64/percpu.zig rename to src/kernel/arch/x86_64/per-cpu.zig index 149e308..4e9e8f2 100644 --- a/src/kernel/arch/x86_64/percpu.zig +++ b/src/kernel/arch/x86_64/per-cpu.zig @@ -1,74 +1,74 @@ //! Per-CPU data reached through the GS segment base. The GS base holds a pointer -//! to this core's `ArchPerCpu`, so kernel code gets the running core's block with -//! a single MSR read (`sched()`) and the syscall entry stub gets its kernel stack +//! to this core's `ArchitecturePerCpu`, so kernel code gets the running core's block with +//! a single MSR read (`scheduler()`) and the system_call entry stub gets its kernel stack //! with a `%gs`-relative load (no usable stack yet at that point). //! //! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds //! the user's own GS (which ring 3 may set freely), and this pointer lives in the //! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the -//! syscall stub and the conditional swapgs in isr_common) brings it back, and +//! system_call stub and the conditional swapgs in isr_common) brings it back, and //! every ring-0 -> ring-3 exit swaps it away. Because the very first ring //! transition is always an exit (the kernel starts in ring 0), the swap pairs -//! keep the invariant without seeding KERNEL_GS_BASE. `sched()` is therefore +//! keep the invariant without seeding KERNEL_GS_BASE. `scheduler()` is therefore //! valid in any ring-0 context and never sees a user-controlled base. const std = @import("std"); const io = @import("io.zig"); -const config = @import("config"); +const parameters = @import("parameters"); const ia32_gs_base = 0xC000_0101; -/// Layout is load-bearing: the syscall entry stub in isr.s reaches `kernel_rsp` +/// Layout is load-bearing: the system_call entry stub in isr.s reaches `kernel_rsp` /// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below /// pin the offsets. -pub const ArchPerCpu = extern struct { - kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for syscall entry (== TSS.rsp0) - scratch: u64 = 0, // %gs:8 — stashes the user rsp during syscall entry - sched: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns) +pub const ArchitecturePerCpu = extern struct { + kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for system_call entry (== TSS.rsp0) + scratch: u64 = 0, // %gs:8 — stashes the user rsp during system_call entry + scheduler: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns) }; comptime { - std.debug.assert(@offsetOf(ArchPerCpu, "kernel_rsp") == 0); - std.debug.assert(@offsetOf(ArchPerCpu, "scratch") == 8); + std.debug.assert(@offsetOf(ArchitecturePerCpu, "kernel_rsp") == 0); + std.debug.assert(@offsetOf(ArchitecturePerCpu, "scratch") == 8); } -var blocks = [_]ArchPerCpu{.{}} ** config.max_cpus; +var blocks = [_]ArchitecturePerCpu{.{}} ** parameters.maximum_cpus; /// Publish core `index`'s per-CPU block: record the scheduler pointer and point /// the GS base at the block. Called once per core during bring-up, after the GDT /// is loaded (a GS *selector* reload would clobber the base). -pub fn setLocal(index: usize, sched_ptr: usize) void { - blocks[index].sched = sched_ptr; +pub fn setLocal(index: usize, scheduler_ptr: usize) void { + blocks[index].scheduler = scheduler_ptr; io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index])); } /// The scheduler pointer for the running core (via the GS base). Valid in any /// ring-0 context under the swapgs discipline. -pub fn sched() usize { - return @as(*const ArchPerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).sched; +pub fn scheduler() usize { + return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler; } -/// Record core `index`'s kernel stack top, used by the syscall entry stub to +/// Record core `index`'s kernel stack top, used by the system_call entry stub to /// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it /// switches to a user task. pub fn setKernelRsp(index: usize, top: usize) void { blocks[index].kernel_rsp = top; } -// Fast-syscall MSRs. +// Fast-system_call MSRs. const ia32_efer = 0xC000_0080; const ia32_star = 0xC000_0081; const ia32_lstar = 0xC000_0082; const ia32_sfmask = 0xC000_0084; -/// Enable the `syscall`/`sysret` fast path on this core (BSP and each AP). EFER.SCE -/// turns the instructions on; STAR sets the selectors syscall/sysret load; LSTAR +/// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE +/// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR /// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF — /// the handler runs with interrupts off, like the int-gate path). The GDT is laid /// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line -/// up: syscall loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8 +/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8 /// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B. -pub fn initSyscall() void { +pub fn initSystemCall() void { io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48)); const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" }); diff --git a/src/kernel/arch/x86_64/serial.zig b/src/kernel/arch/x86_64/serial.zig index 990e903..6b9fbb4 100644 --- a/src/kernel/arch/x86_64/serial.zig +++ b/src/kernel/arch/x86_64/serial.zig @@ -33,13 +33,13 @@ fn portIn(p: u16) u8 { } /// Read UART register `off` through the active access method. -fn reg(off: u64) u8 { +fn register(off: u64) u8 { if (access == .mmio) return @as(*volatile u8, @ptrFromInt(base + off)).*; return portIn(@intCast(base + off)); } /// Write UART register `off` through the active access method. -fn setReg(off: u64, value: u8) void { +fn setRegister(off: u64, value: u8) void { if (access == .mmio) { @as(*volatile u8, @ptrFromInt(base + off)).* = value; } else { @@ -50,22 +50,22 @@ fn setReg(off: u64, value: u8) void { /// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything /// else; it has no dependencies, and is a harmless no-op if the port is absent. pub fn init() void { - setReg(1, 0x00); // disable interrupts - setReg(3, 0x80); // enable DLAB (set baud divisor) - setReg(0, 0x03); // divisor low: 38400 baud - setReg(1, 0x00); // divisor high - setReg(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off - setReg(2, 0xC7); // enable + clear FIFO, 14-byte threshold - setReg(4, 0x0B); // RTS/DSR set + setRegister(1, 0x00); // disable interrupts + setRegister(3, 0x80); // enable DLAB (set baud divisor) + setRegister(0, 0x03); // divisor low: 38400 baud + setRegister(1, 0x00); // divisor high + setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off + setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold + setRegister(4, 0x0B); // RTS/DSR set } /// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and /// re-run the UART setup there. Called after discovery when an SPCR entry exists. -pub fn reconfigure(is_mmio: bool, addr: u64) void { +pub fn reconfigure(is_mmio: bool, address: u64) void { access = if (is_mmio) .mmio else .port; // An MMIO UART is reached through the physmap; an I/O-port UART keeps its // port number unchanged. - base = if (is_mmio) paging.mapMmio(addr, 0x100, true) else addr; + base = if (is_mmio) paging.mapMmio(address, 0x100, true) else address; init(); } @@ -73,8 +73,8 @@ fn writeByte(c: u8) void { // Wait for the transmit-holding register to empty — but bounded, so an absent // UART (whose line-status register reads back as 0x00) can't hang the kernel. var guard: u32 = 0; - while (reg(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {} - setReg(0, c); + while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {} + setRegister(0, c); } /// Write bytes, translating LF to CRLF so terminals and logs line up. diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig index 419764a..14c98de 100644 --- a/src/kernel/arch/x86_64/smp.zig +++ b/src/kernel/arch/x86_64/smp.zig @@ -20,7 +20,7 @@ const tss = @import("tss.zig"); const idt = @import("idt.zig"); const apic = @import("apic.zig"); const paging = @import("paging.zig"); -const pcpu = @import("percpu.zig"); +const pcpu = @import("per-cpu.zig"); /// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP /// path doesn't depend on cpu.zig and risk an import cycle). @@ -31,9 +31,9 @@ const page_size = 0x1000; /// the life of the system so any core can be (re)woken on demand — a retry, or a /// future power manager bringing a core back online. The frame is kept **inert** /// between wakes (zeroed and non-executable) and only armed for the brief moment a -/// core is actually climbing. Its low 20 bits are zero, so `phys >> 12` is the SIPI +/// core is actually climbing. Its low 20 bits are zero, so `physical >> 12` is the SIPI /// vector. -var tramp_phys: u64 = 0; +var tramp_physical: u64 = 0; /// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own /// bring-up. The BSP clears it before each wake and polls it afterwards — a simple @@ -44,7 +44,7 @@ var ap_alive: u32 = 0; /// wake, read by `apEntry` (safe because bring-up is strictly one core at a time). var boot_index: usize = 0; -/// The generic scheduler entry a woken core jumps to once its arch state is up. Set +/// The generic scheduler entry a woken core jumps to once its architecture state is up. Set /// by the kernel via `setSecondaryEntry`; never returns. var secondary_entry: ?*const fn () callconv(.c) noreturn = null; @@ -55,7 +55,7 @@ pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void { /// Test hook: force the next `n` wake attempts to fail (skipping the actual /// INIT-SIPI-SIPI), so the retry path can be exercised deterministically. Zero in -/// normal operation — the smp-retry test arms it via `arch.testFailNextWakes`. +/// normal operation — the smp-retry test arms it via `architecture.testFailNextWakes`. var fail_next_wakes: u32 = 0; pub fn testFailNextWakes(n: u32) void { fail_next_wakes = n; @@ -64,14 +64,14 @@ pub fn testFailNextWakes(n: u32) void { /// Record the reserved low frame the trampoline uses. Call once at boot. The frame /// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms /// it again, so it's only ever executable while a core is climbing. -pub fn setTrampolinePage(phys: u64) void { - tramp_phys = phys; +pub fn setTrampolinePage(physical: u64) void { + tramp_physical = physical; } /// The reserved trampoline frame (0 if SMP bring-up never ran). Exposed so a test /// can verify it's inert — zeroed and non-executable — when dormant. pub fn trampolinePage() u64 { - return tramp_phys; + return tramp_physical; } /// Arm the trampoline for a wake: make its page executable (W^X exception for the @@ -81,12 +81,12 @@ fn arm() void { // climbs from real to long mode, so it needs a low identity mapping that is // executable — the one deliberate, transient W^X exception. The BSP writes // the blob into the frame through the physmap. - paging.setExecutable(tramp_phys); + paging.setExecutable(tramp_physical); const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" }); const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" }); const len = @intFromPtr(end) - @intFromPtr(start); - const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys)); - @memcpy(dst[0..len], start[0..len]); + const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical)); + @memcpy(destination[0..len], start[0..len]); } /// Disarm after a wake: wipe the page through the physmap and remove its low @@ -95,9 +95,9 @@ fn arm() void { /// reported in — it's long past the trampoline by then, in the kernel image; a /// core that never answered is dead and can't be mid-climb. fn disarm() void { - const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys)); - @memset(dst[0..page_size], 0); - paging.unmap(tramp_phys); // drop the transient low identity mapping + const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical)); + @memset(destination[0..page_size], 0); + paging.unmap(tramp_physical); // drop the transient low identity mapping } /// Address of a patchable trampoline parameter, by symbol name: the copied blob's @@ -107,7 +107,7 @@ fn disarm() void { fn param(comptime name: []const u8) *align(1) volatile u64 { const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" })); const sym = @intFromPtr(@extern([*]const u8, .{ .name = name })); - return @ptrFromInt(danos.physToVirt(tramp_phys + (sym - start))); + return @ptrFromInt(danos.physicalToVirtual(tramp_physical + (sym - start))); } /// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it @@ -138,7 +138,7 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: @atomicStore(u32, &ap_alive, 0, .seq_cst); - const vector: u8 = @intCast(tramp_phys >> 12); + const vector: u8 = @intCast(tramp_physical >> 12); apic.sendInit(apic_id); delayMicros(10_000); // 10 ms INIT settle apic.sendStartup(apic_id, vector); @@ -170,12 +170,12 @@ fn apEntry(percpu: usize) callconv(.c) noreturn { tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR idt.loadOnThisCpu(); // the shared IDT pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload - pcpu.initSyscall(); // enable syscall/sysret on this core + pcpu.initSystemCall(); // enable system_call/sysret on this core apic.initSecondary(); // software-enable this core's LAPIC apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off) - @atomicStore(u32, &ap_alive, 1, .release); // "arch state up" — BSP is polling this + @atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop while (true) asm volatile ("hlt"); // (only if no entry was registered) diff --git a/src/kernel/arch/x86_64/tss.zig b/src/kernel/arch/x86_64/tss.zig index 47b8e32..6892c30 100644 --- a/src/kernel/arch/x86_64/tss.zig +++ b/src/kernel/arch/x86_64/tss.zig @@ -11,7 +11,7 @@ //! once can't share one fault stack. So the TSS and its IST stack are per-core, //! indexed by CPU number; slot 0 is the BSP. -const config = @import("config"); +const parameters = @import("parameters"); const gdt = @import("gdt.zig"); /// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned @@ -37,17 +37,17 @@ const Tss = packed struct { /// The IST slot (1-based, as the IDT gate encodes it) used for critical faults. pub const double_fault_ist = 1; -const max_cpus = config.max_cpus; -pub const ist_stack_size = config.ist_stack_size; +const maximum_cpus = parameters.maximum_cpus; +pub const ist_stack_size = parameters.ist_stack_size; /// One TSS per core (small — kept static). The IST stacks are 16 KiB each, so only /// the **BSP's** is static: it must exist before the frame allocator does, to catch a /// fault during early boot. Each **AP** gets a heap-allocated IST stack at bring-up /// (after the heap is up), the top of which the BSP records here before waking it — /// so we reserve big stacks only for cores that actually come online. -var tss_table = [_]Tss{.{}} ** max_cpus; +var tss_table = [_]Tss{.{}} ** maximum_cpus; var bsp_ist_stack: [ist_stack_size]u8 align(16) = undefined; -var ap_ist_top = [_]usize{0} ** max_cpus; // per-AP IST stack top (0 = BSP / not set) +var ap_ist_top = [_]usize{0} ** maximum_cpus; // per-AP IST stack top (0 = BSP / not set) /// Loads the task register with the TSS selector. Defined in isr.s. extern fn load_tr(selector: u16) callconv(.c) void; diff --git a/src/kernel/console.zig b/src/kernel/console.zig index 5cb3d38..05821c3 100644 --- a/src/kernel/console.zig +++ b/src/kernel/console.zig @@ -71,7 +71,7 @@ pub const Console = struct { // once the low identity map is gone. The base is mapped by both the // loader's bootstrap tables and paging.init. var mapped = fb; - if (fb.base != 0) mapped.base = danos.physToVirt(fb.base); + if (fb.base != 0) mapped.base = danos.physicalToVirtual(fb.base); return .{ .fb = mapped, .cols = fb.width / glyph_w, @@ -147,11 +147,11 @@ pub const Console = struct { while (x < self.fb.width) : (x += 1) row[x] = color; } - fn copyRow(self: *Console, dst_y: u32, src_y: u32) void { - const dst = self.rowPtr(dst_y); - const src = self.rowPtr(src_y); + fn copyRow(self: *Console, destination_y: u32, source_y: u32) void { + const destination = self.rowPtr(destination_y); + const source = self.rowPtr(source_y); var x: u32 = 0; - while (x < self.fb.width) : (x += 1) dst[x] = src[x]; + while (x < self.fb.width) : (x += 1) destination[x] = source[x]; } }; diff --git a/src/kernel/device-service.zig b/src/kernel/device-service.zig new file mode 100644 index 0000000..6231478 --- /dev/null +++ b/src/kernel/device-service.zig @@ -0,0 +1,181 @@ +//! Device service: the kernel side of user-space driver access. At boot it +//! flattens the discovered device tree (src/device) into a stable, id-indexed +//! snapshot and a per-device claim table. User drivers enumerate the snapshot, +//! claim the device they own, and map its MMIO — the claim is the capability that +//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the +//! firmware-neutral device tree says it owns. +//! +//! The table is a **tree**: each entry carries its parent's id. Firmware discovery +//! seeds it, and a **bus driver** grows it — a process that has claimed a bus can +//! `register` children below it as it enumerates them (USB devices behind a hub, PCI +//! functions behind a bridge, comparators inside a timer block). +//! +//! Registration is where the capability model earns its keep. A `DeviceDescriptor` is, in +//! effect, a licence to map physical memory: whoever claims it may `mmio_map` its +//! `.memory` resources and `irq_bind` its `.irq` resources. If a bus driver could +//! invent arbitrary resources, it would invent one covering the kernel's RAM, claim +//! it, and map it. So `register` enforces **containment**: every resource of a child +//! must lie inside a resource of the same kind on its parent. A bus driver can only +//! ever subdivide what it was already given. + +const std = @import("std"); +const platform = @import("platform"); +const danos = @import("danos"); + +const maximum_devices = 64; + +/// Cap on children a single parent may have. A zero-resource child (legal — a USB +/// device is addressed through its controller, not by MMIO) sidesteps the containment +/// check, so without a bound a process that claimed one device could loop +/// `device_register` and exhaust the whole table, permanently denying it to every other +/// driver. This bounds the blast radius of one claim; a real quota (and a +/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md. +const maximum_children_per_parent = 16; + +var devices: [maximum_devices]danos.DeviceDescriptor = undefined; +var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null +var count: usize = 0; + +/// Devices discovery found but the table had no room for. Non-zero means the machine +/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers — +/// which would otherwise be an entirely silent failure. Logged at boot. +pub var dropped: usize = 0; + +/// Snapshot the device tree into the flat table. Run once, right after discovery. +pub fn init(device_tree: *const platform.DeviceTree) void { + count = 0; + dropped = 0; + for (&claimed) |*c| c.* = null; + walk(device_tree.root, danos.no_parent); +} + +/// Record `node` (unless it's the synthetic root) and recurse, threading the id we +/// assigned it down to its children as their parent. +fn walk(node: *platform.Device, parent_id: u64) void { + const id = if (node.class == .root) danos.no_parent else record(node, parent_id); + var child = node.first_child; + while (child) |c| : (child = c.next_sibling) walk(c, id); +} + +fn record(node: *platform.Device, parent_id: u64) u64 { + if (count >= maximum_devices) { + dropped += 1; + return danos.no_parent; // children of a dropped node become roots, not orphans + } + var d = std.mem.zeroes(danos.DeviceDescriptor); + d.id = count; + d.parent = parent_id; + d.class = @intFromEnum(node.class); + const h = node.hid(); + d.hid_len = @min(h.len, d.hid.len); + @memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]); + const rc = @min(node.resource_count, danos.maximum_device_resources); + d.resource_count = rc; + for (0..rc) |i| { + const r = node.resources[i]; + d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len }; + } + devices[count] = d; + count += 1; + return d.id; +} + +/// Copy up to `out.len` device descriptors into `out`; returns the total count +/// available (which may exceed `out.len`). +pub fn enumerate(out: []danos.DeviceDescriptor) usize { + const n = @min(count, out.len); + @memcpy(out[0..n], devices[0..n]); + return count; +} + +/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is +/// out of range or already claimed. +pub fn claim(id: u64, owner: u32) bool { + if (id >= count) return false; + if (claimed[@intCast(id)] != null) return false; + claimed[@intCast(id)] = owner; + return true; +} + +/// The task that owns device `id`, or null. +pub fn ownerOf(id: u64) ?u32 { + if (id >= count) return null; + return claimed[@intCast(id)]; +} + +/// Resource `index` of device `id`, or null if out of range. +pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor { + if (id >= count) return null; + const d = &devices[@intCast(id)]; + if (index >= d.resource_count) return null; + return d.resources[@intCast(index)]; +} + +/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's +/// interval containment; for an irq it's equality, since an interrupt line is not +/// divisible. Zero-length child ranges are refused — an empty window is meaningless +/// and would otherwise vacuously "fit" anywhere. +fn contains(parent: danos.ResourceDescriptor, child: danos.ResourceDescriptor) bool { + if (parent.kind != child.kind) return false; + if (child.kind == @intFromEnum(danos.ResourceKind.irq)) return parent.start == child.start; + if (child.len == 0 or parent.len == 0) return false; + // No overflow: a resource that wraps the address space is not containable. + const child_end = std.math.add(u64, child.start, child.len) catch return false; + const parent_end = std.math.add(u64, parent.start, parent.len) catch return false; + return child.start >= parent.start and child_end <= parent_end; +} + +pub const RegisterError = error{ + NoSpace, // the device table is full + BadParent, // no such device, or not claimed by this task + TooManyResources, + TooManyChildren, // this parent is at maximum_children_per_parent + NotContained, // a child resource escapes its parent's window +}; + +/// Number of devices currently recorded with `parent_id` as their parent. +fn childCount(parent_id: u64) usize { + var n: usize = 0; + for (devices[0..count]) |d| { + if (d.parent == parent_id) n += 1; + } + return n; +} + +/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new +/// device id. The child is left **unclaimed**, so another process (a class driver) +/// can claim it — that is how a bus hands a device to its driver. +/// +/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be +/// contained in a parent resource of the same kind. A device with no resources is +/// fine and common: a USB device is addressed through its controller, not by MMIO. +pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescriptor) RegisterError!u64 { + const parent_owner = ownerOf(parent_id) orelse return error.BadParent; + if (parent_owner != owner) return error.BadParent; + if (descriptor.resource_count > danos.maximum_device_resources) return error.TooManyResources; + if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren; + if (count >= maximum_devices) return error.NoSpace; + + const parent = &devices[@intCast(parent_id)]; + for (0..@intCast(descriptor.resource_count)) |i| { + const r = descriptor.resources[i]; + var ok = false; + for (0..@intCast(parent.resource_count)) |j| { + if (contains(parent.resources[j], r)) ok = true; + } + if (!ok) return error.NotContained; + } + + var d = std.mem.zeroes(danos.DeviceDescriptor); + d.id = count; + d.parent = parent_id; + d.class = descriptor.class; + d.hid_len = @min(descriptor.hid_len, d.hid.len); + @memcpy(d.hid[0..@intCast(d.hid_len)], descriptor.hid[0..@intCast(d.hid_len)]); + d.resource_count = descriptor.resource_count; + for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i]; + + devices[count] = d; + count += 1; + return d.id; +} diff --git a/src/kernel/devsvc.zig b/src/kernel/devsvc.zig deleted file mode 100644 index f4fa623..0000000 --- a/src/kernel/devsvc.zig +++ /dev/null @@ -1,78 +0,0 @@ -//! Device service: the kernel side of user-space driver access. At boot it -//! flattens the discovered device tree (src/device) into a stable, id-indexed -//! snapshot and a per-device claim table. User drivers enumerate the snapshot, -//! claim the device they own, and map its MMIO — the claim is the capability that -//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the -//! firmware-neutral device tree says it owns. - -const std = @import("std"); -const platform = @import("platform"); -const danos = @import("danos"); - -const max_devices = 32; - -var devices: [max_devices]danos.DeviceDesc = undefined; -var claimed: [max_devices]?u32 = .{null} ** max_devices; // owner task id, or null -var count: usize = 0; - -/// Snapshot the device tree into the flat table. Run once, right after discovery. -pub fn init(dt: *const platform.DeviceTree) void { - count = 0; - for (&claimed) |*c| c.* = null; - walk(dt.root); -} - -fn walk(node: *platform.Device) void { - if (node.class != .root) record(node); - var child = node.first_child; - while (child) |c| : (child = c.next_sibling) walk(c); -} - -fn record(node: *platform.Device) void { - if (count >= max_devices) return; - var d = std.mem.zeroes(danos.DeviceDesc); - d.id = count; - d.class = @intFromEnum(node.class); - const h = node.hid(); - d.hid_len = @min(h.len, d.hid.len); - @memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]); - const rc = @min(node.resource_count, danos.max_dev_resources); - d.resource_count = rc; - for (0..rc) |i| { - const r = node.resources[i]; - d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len }; - } - devices[count] = d; - count += 1; -} - -/// Copy up to `out.len` device descriptors into `out`; returns the total count -/// available (which may exceed `out.len`). -pub fn enumerate(out: []danos.DeviceDesc) usize { - const n = @min(count, out.len); - @memcpy(out[0..n], devices[0..n]); - return count; -} - -/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is -/// out of range or already claimed. -pub fn claim(id: u64, owner: u32) bool { - if (id >= count) return false; - if (claimed[@intCast(id)] != null) return false; - claimed[@intCast(id)] = owner; - return true; -} - -/// The task that owns device `id`, or null. -pub fn ownerOf(id: u64) ?u32 { - if (id >= count) return null; - return claimed[@intCast(id)]; -} - -/// Resource `idx` of device `id`, or null if out of range. -pub fn resourceOf(id: u64, idx: u64) ?danos.ResDesc { - if (id >= count) return null; - const d = &devices[@intCast(id)]; - if (idx >= d.resource_count) return null; - return d.resources[@intCast(idx)]; -} diff --git a/src/kernel/heap.zig b/src/kernel/heap.zig index 910f4e5..f73f03c 100644 --- a/src/kernel/heap.zig +++ b/src/kernel/heap.zig @@ -2,7 +2,7 @@ //! //! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the //! heap hands out arbitrary byte-sized blocks from a virtual region, growing on -//! demand by mapping fresh frames into it (arch.mapPage) — the first real user of +//! demand by mapping fresh frames into it (architecture.mapPage) — the first real user of //! the VMM (see docs/paging.md). //! //! The algorithm is a first-fit free list: an address-ordered singly linked list @@ -14,17 +14,17 @@ const std = @import("std"); const danos = @import("danos"); -const arch = @import("arch"); +const architecture = @import("architecture"); const pmm = @import("pmm.zig"); const page_size = danos.page_size; /// Virtual base of the heap: the start of the higher half, which is unmapped and -/// well clear of the identity-mapped low half. (Canonical on x86_64; an arch that +/// well clear of the identity-mapped low half. (Canonical on x86_64; an architecture that /// splits the address space differently would choose its own.) const heap_base: usize = 0xFFFF_8000_0000_0000; /// Cap on heap growth for now. -const heap_max: usize = 64 * 1024 * 1024; +const heap_maximum: usize = 64 * 1024 * 1024; /// A block header, placed at the start of every block. While the block is free it /// also links into the free list via `next`. @@ -34,7 +34,7 @@ const Block = extern struct { }; const header_size = @sizeOf(Block); // 16 -const min_block = header_size + 16; // smallest block worth splitting off +const minimum_block = header_size + 16; // smallest block worth splitting off var free_list: ?*Block = null; var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped @@ -56,15 +56,15 @@ pub fn init() void { /// Map more pages onto the end of the heap and add them as a free block. Returns /// false if out of heap virtual space or out of physical frames. -fn grow(min_bytes: usize) bool { +fn grow(minimum_bytes: usize) bool { const start = heap_end; - const bytes = alignUp(min_bytes, page_size); - if (start + bytes > heap_base + heap_max) return false; + const bytes = alignUp(minimum_bytes, page_size); + if (start + bytes > heap_base + heap_maximum) return false; - var virt = start; - while (virt < start + bytes) : (virt += page_size) { + var virtual = start; + while (virtual < start + bytes) : (virtual += page_size) { const frame = pmm.alloc() orelse return false; - arch.mapPage(virt, frame, true); + architecture.mapPage(virtual, frame, true); } heap_end = start + bytes; @@ -77,25 +77,25 @@ fn grow(min_bytes: usize) bool { /// Insert a block into the address-ordered free list, coalescing with the /// physically adjacent free blocks on either side. fn insertFree(block: *Block) void { - var prev: ?*Block = null; - var cur = free_list; - while (cur) |c| : (cur = c.next) { + var previous: ?*Block = null; + var current = free_list; + while (current) |c| : (current = c.next) { if (@intFromPtr(c) > @intFromPtr(block)) break; - prev = c; + previous = c; } - block.next = cur; - if (prev) |p| p.next = block else free_list = block; + block.next = current; + if (previous) |p| p.next = block else free_list = block; - // Merge forward into `cur` if they're contiguous. - if (cur) |c| { + // Merge forward into `current` if they're contiguous. + if (current) |c| { if (@intFromPtr(block) + block.size == @intFromPtr(c)) { block.size += c.size; block.next = c.next; } } - // Merge `prev` forward into `block` if they're contiguous. - if (prev) |p| { + // Merge `previous` forward into `block` if they're contiguous. + if (previous) |p| { if (@intFromPtr(p) + p.size == @intFromPtr(block)) { p.size += block.size; p.next = block.next; @@ -109,24 +109,24 @@ fn rawAlloc(len: usize) ?[*]u8 { var attempts: u32 = 0; while (attempts < 2) : (attempts += 1) { - var prev: ?*Block = null; - var cur = free_list; - while (cur) |block| : ({ - prev = block; - cur = block.next; + var previous: ?*Block = null; + var current = free_list; + while (current) |block| : ({ + previous = block; + current = block.next; }) { if (block.size < need) continue; - if (block.size >= need + min_block) { + if (block.size >= need + minimum_block) { // Split: carve `need` off the front, leave the rest free. const rest: *Block = @ptrFromInt(@intFromPtr(block) + need); rest.size = block.size - need; rest.next = block.next; - if (prev) |p| p.next = rest else free_list = rest; + if (previous) |p| p.next = rest else free_list = rest; block.size = need; } else { // Take the whole block. - if (prev) |p| p.next = block.next else free_list = block.next; + if (previous) |p| p.next = block.next else free_list = block.next; } return payloadOf(block); } diff --git a/src/kernel/ipc-synchronous.zig b/src/kernel/ipc-synchronous.zig new file mode 100644 index 0000000..dcce2e5 --- /dev/null +++ b/src/kernel/ipc-synchronous.zig @@ -0,0 +1,314 @@ +//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a +//! rendezvous point; a client `call`s it (send a message, block for a reply) and +//! a server `replyWait`s on it (reply to the last client, then block for the next +//! request). This is the substrate the user-space VFS server and device drivers +//! are reached through — `open`/`read`/`write` become user-space wrappers that +//! marshal a request into a `call`. +//! +//! Design (see docs/syscall.md, the plan): +//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame +//! through the physmap (`copyAcross`), which is mapped in every address space's +//! shared kernel half — so the kernel reads/writes either process's user memory +//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing. +//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply +//! to exactly one client at a time; that caller is held in `Task.ipc_client`. +//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without +//! becoming runnable*, which a WaitQueue can't express, so callers queue on the +//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers +//! waiting for work use a normal WaitQueue. +//! +//! Trust model (bring-up): copies honour only page presence and a user-half bound, +//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked +//! copy is a later security-track item, matching the existing debug_write gap. + +const std = @import("std"); +const danos = @import("danos"); +const architecture = @import("architecture"); +const scheduler = @import("scheduler.zig"); +const sync = @import("sync.zig"); +const heap = @import("heap.zig"); + +const page_size = danos.page_size; +const Task = scheduler.Task; + +/// Largest message a single call/reply may carry. Bumping it is trivial; kept +/// small because the copy runs under the big kernel lock. +pub const MESSAGE_MAXIMUM: usize = 256; + +pub const maximum_handles = scheduler.ipc_maximum_handles; +pub const maximum_services = 8; + +/// Errno-style failures, returned as `-value` in the system_call result register. +pub const EBADF: i64 = 1; // bad handle +pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM +pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half +pub const ENOENT: i64 = 4; // no such registered service +pub const ENOSPC: i64 = 5; // handle table or registry full +pub const ENOMEM: i64 = 6; // out of memory + +/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a +/// message from a client — there is no reply owed. The low bits carry the source +/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in src/kernel/irq.zig; +/// the message path uses a plain task-id badge with this bit clear. Defined in the +/// shared contract (src/root.zig), because ring 3 has to test the same bit. +pub const notify_badge_bit: u64 = danos.notify_badge_bit; + +/// End of the user (low) canonical half — user buffers must lie below it. +const user_half_end: u64 = 0x0000_8000_0000_0000; + +/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle +/// (per process) and/or by a registry slot, counted by `refcount`. +pub const Endpoint = struct { + refcount: u32 = 1, + // Callers blocked in `call`, awaiting receive, in FIFO order (threaded via + // Task.next; each such task is .blocked and in no scheduler queue). + sender_head: ?*Task = null, + sender_tail: ?*Task = null, + // Servers blocked in `replyWait` awaiting a request. + receive_wait_queue: scheduler.WaitQueue = .{}, + // Pending asynchronous notifications (badges), a small coalescing ring. + notify_buffer: [8]u64 = undefined, + notify_head: u8 = 0, + notify_tail: u8 = 0, +}; + +pub fn createEndpoint() ?*Endpoint { + const endpoint = heap.allocator().create(Endpoint) catch return null; + endpoint.* = .{}; + return endpoint; +} + +/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked +/// today like other kernel objects — but the refcount bookkeeping lands now.) +pub fn dropRef(endpoint: *Endpoint) void { + if (endpoint.refcount > 1) { + endpoint.refcount -= 1; + } else { + heap.allocator().destroy(endpoint); + } +} + +// --- sender FIFO (endpoint-local, via Task.next) ---------------------------- + +fn enqueueSender(endpoint: *Endpoint, t: *Task) void { + t.next = null; + if (endpoint.sender_tail) |tail| tail.next = t else endpoint.sender_head = t; + endpoint.sender_tail = t; +} + +fn dequeueSender(endpoint: *Endpoint) ?*Task { + const t = endpoint.sender_head orelse return null; + endpoint.sender_head = t.next; + if (endpoint.sender_head == null) endpoint.sender_tail = null; + t.next = null; + return t; +} + +// --- cross-address-space copy ---------------------------------------------- + +/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in +/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch). +/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User +/// buffers must lie in the low half. Returns false — never #PFs — if any page is +/// unmapped or out of range. Handles page-straddling buffers. +fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool { + const source_root = if (source_as != 0) source_as else architecture.kernelPageTable(); + const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable(); + if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false; + if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false; + + var off: usize = 0; + while (off < len) { + const s = architecture.translate(source_root, source_va + off) orelse return false; + const d = architecture.translate(destination_root, destination_va + off) orelse return false; + const s_left = page_size - ((source_va + off) & (page_size - 1)); + const d_left = page_size - ((destination_va + off) & (page_size - 1)); + const n = @min(@min(s_left, d_left), len - off); + const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s)); + const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(d)); + @memcpy(destination[0..n], source[0..n]); + off += n; + } + return true; +} + +/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel +/// buffer `destination`, walking the user page tables through the physmap. Returns false if +/// the range escapes the user half or any source page is unmapped — so a bad user +/// pointer *fails the system_call* rather than faulting the kernel (danos has no +/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt +/// the machine). The correct way to pull a fixed-size struct in from user space, and +/// a single fetch: no TOCTOU against a hostile pointer. +pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool { + if (user_as == 0) return false; // not a user address space + if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false; + var off: usize = 0; + while (off < destination.len) { + const s = architecture.translate(user_as, user_va + off) orelse return false; + const s_left = page_size - ((user_va + off) & (page_size - 1)); + const n = @min(s_left, destination.len - off); + const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s)); + @memcpy(destination[off..][0..n], source[0..n]); + off += n; + } + return true; +} + +// --- the two IPC operations ------------------------------------------------- + +/// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a +/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a +/// negative errno. Runs as the current task. +pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 { + if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG; + const flags = sync.enter(); + defer sync.leave(flags); + + const me = scheduler.current(); + me.ipc_send_ptr = message_ptr; + me.ipc_send_len = message_len; + me.ipc_reply_ptr = reply_ptr; + me.ipc_reply_cap = reply_cap; + me.ipc_status = 0; + + enqueueSender(endpoint, me); // join the FIFO, then... + scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none) + scheduler.blockCurrentLocked(); // block until the reply readies us again + + return me.ipc_status; // reply length or -errno, written by the replier +} + +/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client +/// we currently owe (if any), then receive the next request into +/// `[receive_ptr, receive_cap)`, blocking until one arrives. Writes the sender's badge +/// to `out_badge` and returns the request length, or a negative errno. A pending +/// notification is delivered ahead of client requests (length 0, badge with +/// `notify_badge_bit` set, no reply owed). +pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 { + if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG; + const flags = sync.enter(); + defer sync.leave(flags); + + const me = scheduler.current(); + + // (1) Reply to the client we're still holding, if any. + if (me.ipc_client) |client| { + me.ipc_client = null; + const n = @min(reply_len, client.ipc_reply_cap); + if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { + client.ipc_status = @intCast(n); + } else { + client.ipc_status = -EFAULT; + } + scheduler.readyLocked(client); // its `call` now returns + } + + // (2) Receive the next request (or notification), blocking until one is ready. + while (true) { + if (popNotify(endpoint)) |badge| { + out_badge.* = badge | notify_badge_bit; + return 0; // notification: no payload, no reply owed + } + if (dequeueSender(endpoint)) |caller| { + const n = @min(caller.ipc_send_len, receive_cap); + if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) { + caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving + scheduler.readyLocked(caller); + continue; + } + me.ipc_client = caller; // remember who to reply to + out_badge.* = caller.id; + return @intCast(n); + } + scheduler.waitLocked(&endpoint.receive_wait_queue); // nothing yet — sleep until woken, then retry + } +} + +// --- asynchronous notification (for IRQ-as-message, M10) -------------------- + +fn popNotify(endpoint: *Endpoint) ?u64 { + if (endpoint.notify_head == endpoint.notify_tail) return null; + const badge = endpoint.notify_buffer[endpoint.notify_head % endpoint.notify_buffer.len]; + endpoint.notify_head +%= 1; + return badge; +} + +/// Post an asynchronous notification carrying `badge` to `endpoint` and wake a waiting +/// receiver. Precondition: the big kernel lock is held. +/// +/// The lock must already cover whatever produced `endpoint` — an ISR that looked the +/// endpoint up in a table and *then* took the lock could be racing a process exit +/// that unbinds and frees it in between. See irq.dispatch, which holds one lock +/// region across the table read and this call. +/// +/// A full ring drops the notification. That is the correct semantics, not a +/// concession: a notification is a *level* ("this device wants attention"), and the +/// driver re-reads device state on wake. It is never a count of events. +pub fn notifyLocked(endpoint: *Endpoint, badge: u64) void { + if (endpoint.notify_tail -% endpoint.notify_head < endpoint.notify_buffer.len) { + endpoint.notify_buffer[endpoint.notify_tail % endpoint.notify_buffer.len] = badge; + endpoint.notify_tail +%= 1; + } + scheduler.wakeLocked(&endpoint.receive_wait_queue); +} + +/// `notifyLocked` as a self-contained ISR critical section, for a caller that holds +/// `endpoint` by some means other than a table the lock protects. Releases the lock without +/// touching the interrupt flag (the ISR's iretq restores it), like the timer tick. +pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void { + _ = sync.enter(); + notifyLocked(endpoint, badge); + sync.leaveIsr(); +} + +// --- per-process handle table + name registry ------------------------------- + +/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or +/// -ENOSPC. The caller has already taken/holds the reference the slot represents. +pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 { + for (&t.handles, 0..) |*slot, i| { + if (slot.* == null) { + slot.* = @ptrCast(endpoint); + return @intCast(i); + } + } + return -ENOSPC; +} + +/// Resolve a handle to its endpoint, or null if out of range / unused. +pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint { + if (h >= t.handles.len) return null; + const slot = t.handles[@intCast(h)] orelse return null; + return @ptrCast(@alignCast(slot)); +} + +/// Drop every endpoint reference an exiting task holds. Called from the scheduler +/// exit path so a dead server's endpoints don't linger referenced. +pub fn closeHandles(t: *Task) void { + for (&t.handles) |*slot| { + if (slot.*) |p| { + dropRef(@ptrCast(@alignCast(p))); + slot.* = null; + } + } +} + +var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services; + +/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno. +pub fn register(id: u32, endpoint: *Endpoint) i64 { + if (id >= maximum_services) return -ENOENT; + if (registry[id]) |old| dropRef(old); + endpoint.refcount += 1; + registry[id] = endpoint; + return 0; +} + +/// Find the endpoint published under `id`, taking a reference for the caller to +/// install in its handle table. Null if nothing is registered there. +pub fn lookup(id: u32) ?*Endpoint { + if (id >= maximum_services) return null; + const endpoint = registry[id] orelse return null; + endpoint.refcount += 1; + return endpoint; +} diff --git a/src/kernel/ipc.zig b/src/kernel/ipc.zig index 34de07c..9656e6e 100644 --- a/src/kernel/ipc.zig +++ b/src/kernel/ipc.zig @@ -4,14 +4,14 @@ //! drivers and services live in separate address spaces, a message is how they //! talk. This first form is a **bounded blocking channel** — a ring buffer of //! messages with a producer/consumer rendezvous, built on the scheduler's -//! [wait queues](scheduling.md). `send` blocks when the channel is full, `recv` +//! [wait queues](scheduling.md). `send` blocks when the channel is full, `receive` //! blocks when it's empty; neither busy-waits. //! //! For now both endpoints are kernel threads sharing the kernel address space. //! When user mode arrives, the same primitive carries messages across the //! isolation boundary (with the payload copied between address spaces). -const sched = @import("scheduler.zig"); +const scheduler = @import("scheduler.zig"); const sync = @import("sync.zig"); /// A bounded blocking channel of `capacity` messages of type `T`. @@ -23,32 +23,32 @@ pub fn Channel(comptime T: type, comptime capacity: usize) type { head: usize = 0, // next slot to read tail: usize = 0, // next slot to write count: usize = 0, - not_full: sched.WaitQueue = .{}, // senders wait here - not_empty: sched.WaitQueue = .{}, // receivers wait here + not_full: scheduler.WaitQueue = .{}, // senders wait here + not_empty: scheduler.WaitQueue = .{}, // receivers wait here /// Send a message, blocking while the channel is full. - pub fn send(self: *Self, msg: T) void { + pub fn send(self: *Self, message: T) void { const flags = sync.enter(); // Recheck the condition in a loop: a wakeup only means "try again" // (another waiter may have taken the slot first). - while (self.count == capacity) sched.waitLocked(&self.not_full); - self.buffer[self.tail] = msg; + while (self.count == capacity) scheduler.waitLocked(&self.not_full); + self.buffer[self.tail] = message; self.tail = (self.tail + 1) % capacity; self.count += 1; - sched.wakeLocked(&self.not_empty); // a receiver can now proceed + scheduler.wakeLocked(&self.not_empty); // a receiver can now proceed sync.leave(flags); } /// Receive a message, blocking while the channel is empty. - pub fn recv(self: *Self) T { + pub fn receive(self: *Self) T { const flags = sync.enter(); - while (self.count == 0) sched.waitLocked(&self.not_empty); - const msg = self.buffer[self.head]; + while (self.count == 0) scheduler.waitLocked(&self.not_empty); + const message = self.buffer[self.head]; self.head = (self.head + 1) % capacity; self.count -= 1; - sched.wakeLocked(&self.not_full); // a sender can now proceed + scheduler.wakeLocked(&self.not_full); // a sender can now proceed sync.leave(flags); - return msg; + return message; } }; } diff --git a/src/kernel/ipc_sync.zig b/src/kernel/ipc_sync.zig deleted file mode 100644 index 320c5be..0000000 --- a/src/kernel/ipc_sync.zig +++ /dev/null @@ -1,277 +0,0 @@ -//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a -//! rendezvous point; a client `call`s it (send a message, block for a reply) and -//! a server `replyWait`s on it (reply to the last client, then block for the next -//! request). This is the substrate the user-space VFS server and device drivers -//! are reached through — `open`/`read`/`write` become user-space wrappers that -//! marshal a request into a `call`. -//! -//! Design (see docs/syscall.md, the plan): -//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame -//! through the physmap (`copyAcross`), which is mapped in every address space's -//! shared kernel half — so the kernel reads/writes either process's user memory -//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing. -//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply -//! to exactly one client at a time; that caller is held in `Task.ipc_client`. -//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without -//! becoming runnable*, which a WaitQueue can't express, so callers queue on the -//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers -//! waiting for work use a normal WaitQueue. -//! -//! Trust model (bring-up): copies honour only page presence and a user-half bound, -//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked -//! copy is a later security-track item, matching the existing debug_write gap. - -const std = @import("std"); -const danos = @import("danos"); -const arch = @import("arch"); -const sched = @import("scheduler.zig"); -const sync = @import("sync.zig"); -const heap = @import("heap.zig"); - -const page_size = danos.page_size; -const Task = sched.Task; - -/// Largest message a single call/reply may carry. Bumping it is trivial; kept -/// small because the copy runs under the big kernel lock. -pub const MSG_MAX: usize = 256; - -pub const max_handles = sched.ipc_max_handles; -pub const max_services = 8; - -/// Errno-style failures, returned as `-value` in the syscall result register. -pub const EBADF: i64 = 1; // bad handle -pub const E2BIG: i64 = 2; // message exceeds MSG_MAX -pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half -pub const ENOENT: i64 = 4; // no such registered service -pub const ENOSPC: i64 = 5; // handle table or registry full -pub const ENOMEM: i64 = 6; // out of memory - -/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a -/// message from a client — there is no reply owed. The low bits carry the source -/// (a GSI for IRQs). Used by `notifyFromIsr`/M10; the message path uses a plain -/// task-id badge with this bit clear. -pub const notify_badge_bit: u64 = 1 << 63; - -/// End of the user (low) canonical half — user buffers must lie below it. -const user_half_end: u64 = 0x0000_8000_0000_0000; - -/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle -/// (per process) and/or by a registry slot, counted by `refcount`. -pub const Endpoint = struct { - refcount: u32 = 1, - // Callers blocked in `call`, awaiting receive, in FIFO order (threaded via - // Task.next; each such task is .blocked and in no scheduler queue). - sender_head: ?*Task = null, - sender_tail: ?*Task = null, - // Servers blocked in `replyWait` awaiting a request. - recv_wq: sched.WaitQueue = .{}, - // Pending asynchronous notifications (badges), a small coalescing ring. - notify_buf: [8]u64 = undefined, - notify_head: u8 = 0, - notify_tail: u8 = 0, -}; - -pub fn createEndpoint() ?*Endpoint { - const ep = heap.allocator().create(Endpoint) catch return null; - ep.* = .{}; - return ep; -} - -/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked -/// today like other kernel objects — but the refcount bookkeeping lands now.) -pub fn dropRef(ep: *Endpoint) void { - if (ep.refcount > 1) { - ep.refcount -= 1; - } else { - heap.allocator().destroy(ep); - } -} - -// --- sender FIFO (endpoint-local, via Task.next) ---------------------------- - -fn enqueueSender(ep: *Endpoint, t: *Task) void { - t.next = null; - if (ep.sender_tail) |tail| tail.next = t else ep.sender_head = t; - ep.sender_tail = t; -} - -fn dequeueSender(ep: *Endpoint) ?*Task { - const t = ep.sender_head orelse return null; - ep.sender_head = t.next; - if (ep.sender_head == null) ep.sender_tail = null; - t.next = null; - return t; -} - -// --- cross-address-space copy ---------------------------------------------- - -/// Copy `len` bytes from `src_va` in address space `src_as` to `dst_va` in -/// `dst_as`, walking each side's page tables through the physmap (no CR3 switch). -/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User -/// buffers must lie in the low half. Returns false — never #PFs — if any page is -/// unmapped or out of range. Handles page-straddling buffers. -fn copyAcross(src_as: u64, src_va: u64, dst_as: u64, dst_va: u64, len: usize) bool { - const src_root = if (src_as != 0) src_as else arch.kernelPageTable(); - const dst_root = if (dst_as != 0) dst_as else arch.kernelPageTable(); - if (src_as != 0 and (src_va >= user_half_end or src_va + len > user_half_end)) return false; - if (dst_as != 0 and (dst_va >= user_half_end or dst_va + len > user_half_end)) return false; - - var off: usize = 0; - while (off < len) { - const s = arch.translate(src_root, src_va + off) orelse return false; - const d = arch.translate(dst_root, dst_va + off) orelse return false; - const s_left = page_size - ((src_va + off) & (page_size - 1)); - const d_left = page_size - ((dst_va + off) & (page_size - 1)); - const n = @min(@min(s_left, d_left), len - off); - const src: [*]const u8 = @ptrFromInt(danos.physToVirt(s)); - const dst: [*]u8 = @ptrFromInt(danos.physToVirt(d)); - @memcpy(dst[0..n], src[0..n]); - off += n; - } - return true; -} - -// --- the two IPC operations ------------------------------------------------- - -/// Client side of IPC_Call: send `[msg_ptr, msg_len)` to `ep` and block until a -/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a -/// negative errno. Runs as the current task. -pub fn call(ep: *Endpoint, msg_ptr: u64, msg_len: u64, reply_ptr: u64, reply_cap: u64) i64 { - if (msg_len > MSG_MAX or reply_cap > MSG_MAX) return -E2BIG; - const flags = sync.enter(); - defer sync.leave(flags); - - const me = sched.cur(); - me.ipc_send_ptr = msg_ptr; - me.ipc_send_len = msg_len; - me.ipc_reply_ptr = reply_ptr; - me.ipc_reply_cap = reply_cap; - me.ipc_status = 0; - - enqueueSender(ep, me); // join the FIFO, then... - sched.wakeLocked(&ep.recv_wq); // ...wake a waiting server (no-op if none) - sched.blockCurrentLocked(); // block until the reply readies us again - - return me.ipc_status; // reply length or -errno, written by the replier -} - -/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client -/// we currently owe (if any), then receive the next request into -/// `[recv_ptr, recv_cap)`, blocking until one arrives. Writes the sender's badge -/// to `out_badge` and returns the request length, or a negative errno. A pending -/// notification is delivered ahead of client requests (length 0, badge with -/// `notify_badge_bit` set, no reply owed). -pub fn replyWait(ep: *Endpoint, reply_ptr: u64, reply_len: u64, recv_ptr: u64, recv_cap: u64, out_badge: *u64) i64 { - if (reply_len > MSG_MAX or recv_cap > MSG_MAX) return -E2BIG; - const flags = sync.enter(); - defer sync.leave(flags); - - const me = sched.cur(); - - // (1) Reply to the client we're still holding, if any. - if (me.ipc_client) |client| { - me.ipc_client = null; - const n = @min(reply_len, client.ipc_reply_cap); - if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { - client.ipc_status = @intCast(n); - } else { - client.ipc_status = -EFAULT; - } - sched.readyLocked(client); // its `call` now returns - } - - // (2) Receive the next request (or notification), blocking until one is ready. - while (true) { - if (popNotify(ep)) |badge| { - out_badge.* = badge | notify_badge_bit; - return 0; // notification: no payload, no reply owed - } - if (dequeueSender(ep)) |caller| { - const n = @min(caller.ipc_send_len, recv_cap); - if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, recv_ptr, n)) { - caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving - sched.readyLocked(caller); - continue; - } - me.ipc_client = caller; // remember who to reply to - out_badge.* = caller.id; - return @intCast(n); - } - sched.waitLocked(&ep.recv_wq); // nothing yet — sleep until woken, then retry - } -} - -// --- asynchronous notification (for IRQ-as-message, M10) -------------------- - -fn popNotify(ep: *Endpoint) ?u64 { - if (ep.notify_head == ep.notify_tail) return null; - const badge = ep.notify_buf[ep.notify_head % ep.notify_buf.len]; - ep.notify_head +%= 1; - return badge; -} - -/// Post an asynchronous notification carrying `badge` to `ep` and wake a waiting -/// receiver. ISR-safe: takes the big kernel lock and releases it without touching -/// the interrupt flag (the ISR's iretq restores it), exactly like the timer tick. -/// A full ring drops the notification (the driver re-reads device state anyway). -pub fn notifyFromIsr(ep: *Endpoint, badge: u64) void { - _ = sync.enter(); - if (ep.notify_tail -% ep.notify_head < ep.notify_buf.len) { - ep.notify_buf[ep.notify_tail % ep.notify_buf.len] = badge; - ep.notify_tail +%= 1; - } - sched.wakeLocked(&ep.recv_wq); - sync.leaveIsr(); -} - -// --- per-process handle table + name registry ------------------------------- - -/// Install `ep` in task `t`'s handle table; returns the small-int handle or -/// -ENOSPC. The caller has already taken/holds the reference the slot represents. -pub fn installHandle(t: *Task, ep: *Endpoint) i64 { - for (&t.handles, 0..) |*slot, i| { - if (slot.* == null) { - slot.* = @ptrCast(ep); - return @intCast(i); - } - } - return -ENOSPC; -} - -/// Resolve a handle to its endpoint, or null if out of range / unused. -pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint { - if (h >= t.handles.len) return null; - const slot = t.handles[@intCast(h)] orelse return null; - return @ptrCast(@alignCast(slot)); -} - -/// Drop every endpoint reference an exiting task holds. Called from the scheduler -/// exit path so a dead server's endpoints don't linger referenced. -pub fn closeHandles(t: *Task) void { - for (&t.handles) |*slot| { - if (slot.*) |p| { - dropRef(@ptrCast(@alignCast(p))); - slot.* = null; - } - } -} - -var registry: [max_services]?*Endpoint = .{null} ** max_services; - -/// Publish `ep` under well-known `id` (takes a reference). Returns 0 or -errno. -pub fn register(id: u32, ep: *Endpoint) i64 { - if (id >= max_services) return -ENOENT; - if (registry[id]) |old| dropRef(old); - ep.refcount += 1; - registry[id] = ep; - return 0; -} - -/// Find the endpoint published under `id`, taking a reference for the caller to -/// install in its handle table. Null if nothing is registered there. -pub fn lookup(id: u32) ?*Endpoint { - if (id >= max_services) return null; - const ep = registry[id] orelse return null; - ep.refcount += 1; - return ep; -} diff --git a/src/kernel/irq.zig b/src/kernel/irq.zig new file mode 100644 index 0000000..743c7c5 --- /dev/null +++ b/src/kernel/irq.zig @@ -0,0 +1,190 @@ +//! IRQ-as-IPC: delivering a hardware interrupt to a user-space driver. +//! +//! A microkernel can't run driver code in the ISR — the driver is a ring-3 process +//! in another address space. So the kernel's ISR does the least it can: quiet the +//! line, acknowledge the CPU, and post an asynchronous notification to the endpoint +//! the driver is blocked on (`ipc_sync.notifyFromIsr`). The driver wakes out of +//! `IPC_ReplyWait`, services the device, and calls `irq_ack` to re-arm. +//! +//! The full cycle, and why each step is where it is: +//! +//! ISR irqMask(gsi) -- the line is still asserted; stop it reaching a CPU +//! irqEoi() -- now safe to tell the LAPIC we're done +//! notifyFromIsr() -- wake the driver (it runs much later) +//! driver -- reads/clears the device's status register +//! driver irq_ack(device,resource) -- irqUnmask(gsi): the line is quiet, let it through +//! +//! Mask-before-EOI is the load-bearing part. A level-triggered line stays asserted +//! until the *device* is quieted, which only the ring-3 driver can do. EOI with the +//! entry unmasked and the I/O APIC redelivers immediately, forever, before the +//! driver is ever scheduled. Masking converts "level" into something a deferred +//! handler can cope with; `irq_ack` is what closes the loop. +//! +//! Binding is capability-gated exactly like `mmio_map`: the caller must have +//! `device_claim`ed the device, and the GSI must come from one of that device's `irq` +//! resources in the discovered device table (src/kernel/device-service.zig). A driver can +//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let +//! any process steal the keyboard's line. +//! +//! KNOWN ISSUE (real hardware, not QEMU). Masking a level-triggered redirection entry +//! while its remote-IRR bit is set does not clear remote-IRR on some chipsets, and the +//! line then never fires again — a driver would take exactly one interrupt and block +//! forever. QEMU's I/O APIC clears remote-IRR on EOI regardless of the mask, so the +//! `hpet` test cannot see this. Linux's workaround is to flush remote-IRR by briefly +//! flipping the entry to edge-trigger and back. Revisit when danos first boots on +//! metal; MSI (no mask cycle at all) sidesteps it entirely. + +const architecture = @import("architecture"); +const sync = @import("sync.zig"); +const ipc_sync = @import("ipc-synchronous.zig"); + +/// GSIs a single I/O APIC covers. 24 is the standard redirection-table size; a +/// second I/O APIC (none on QEMU's q35) would extend this. +pub const maximum_gsi = 24; + +/// The endpoint to notify for each bound GSI, or null if unbound. Read from the ISR +/// and written from syscalls, always under the big kernel lock. +var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi; + +/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint +/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand +/// out extra references), so "every GSI pointing at this endpoint" is not the same +/// set as "every GSI this process bound", and releasing the former on exit would mask +/// a live sibling's device line. +var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi; + +/// No GSI assigned to this vector. +const no_gsi: u32 = 0xFFFF_FFFF; + +/// Reverse map for the ISR: which GSI does this vector carry? Populated at bind. +/// `interruptDispatch` hands a handler no arguments, so the vector→GSI edge has to +/// be recovered from somewhere — the trampolines below capture the vector at +/// comptime, and this turns it back into a GSI. Trampolines are installed on every +/// vector in the window at boot, so an unbound one must be distinguishable from +/// GSI 0 — hence the sentinel rather than a zero default. +var vector_gsi: [256]u32 = .{no_gsi} ** 256; + +/// Vector currently assigned to each GSI (0 = none), so a rebind reuses it. +var gsi_vector: [maximum_gsi]u8 = .{0} ** maximum_gsi; + +/// Set once the trampolines are installed. +var installed = false; + +/// The ISR body for a bound device line. Runs with interrupts off, on the +/// interrupted task's kernel stack, on whichever core the I/O APIC picked. +fn dispatch(vector: u8) void { + const gsi = vector_gsi[vector]; + if (gsi == no_gsi) { + // Nothing is routed here. Acknowledge so the LAPIC doesn't wedge on an + // in-service bit that never clears, but touch no redirection entry. + architecture.irqEoi(); + return; + } + + // One lock region for the whole cycle. Two reasons, and the second is subtle: + // + // - The I/O APIC is an index/data register pair, so two cores interleaving a + // read-modify-write of a redirection entry would corrupt it. + // - `bound[gsi]` must be *read and used* under the same acquisition that + // `unbind` writes it under. Dropping the lock between the load and + // `notifyLocked` would let a driver exiting on another core free the endpoint + // in the gap, and we would post a notification into freed memory. The GSI is + // routed to the core that bound it, but a driver may migrate and exit + // elsewhere, so this is reachable on SMP. + // + // No deadlock: the lock is non-recursive, but a core holding it runs with + // interrupts disabled and so cannot interrupt itself into here. + _ = sync.enter(); + defer sync.leaveIsr(); + + architecture.irqMask(gsi); // the line is still asserted; stop it reaching a CPU + architecture.irqEoi(); // now safe to release the LAPIC's in-service bit + + // Wakes the driver if it's blocked in ReplyWait; otherwise queues the badge on + // the endpoint's notify ring, so an interrupt taken while the driver is off + // doing something else is not lost. + if (bound[gsi]) |endpoint| ipc_sync.notifyLocked(endpoint, gsi); +} + +/// Install one no-argument trampoline per usable vector. Each closes over its own +/// `vector` as a comptime constant — that's the trick that gets an argument into +/// `idt.Handler` (`*const fn () void`) without a per-vector hand-written stub. +pub fn init() void { + if (installed) return; + inline for (0..architecture.irq_vector_count) |i| { + const vector: u8 = @intCast(@as(usize, architecture.irq_vector_base) + i); + architecture.irqSetHandler(vector, &struct { + fn trampoline() void { + dispatch(vector); + } + }.trampoline); + } + installed = true; +} + +/// Lowest unused vector in the device window, or null if they're all spoken for. +fn allocVector() ?u8 { + var v: u8 = architecture.irq_vector_base; + while (v < architecture.irq_vector_base + architecture.irq_vector_count) : (v += 1) { + var used = false; + for (gsi_vector) |gv| { + if (gv == v) used = true; + } + if (!used) return v; + } + return null; +} + +pub const BindError = error{ BadGsi, InUse, NoVector }; + +/// Deliver `gsi` to `endpoint` as an IPC notification, on behalf of task `owner`. Routes the +/// line to this core, installs the binding, and unmasks. Caller must hold the big +/// kernel lock, and must already have checked that `owner` claimed the device this GSI +/// belongs to. +pub fn bind(gsi: u32, endpoint: *ipc_sync.Endpoint, owner: u32) BindError!void { + if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi; + if (bound[gsi] != null) return error.InUse; + const vector = allocVector() orelse return error.NoVector; + + vector_gsi[vector] = gsi; + gsi_vector[gsi] = vector; + bound[gsi] = endpoint; + bound_owner[gsi] = owner; + + // Level-triggered, active-high. Level is the general case a driver must survive + // (and what hpetd configures its comparator for); an edge source simply never + // leaves the line asserted, so the mask/ack cycle is harmless there. + // + // Hardcoded for now: a device whose MADT interrupt-source override declares the + // line active-*low* (most legacy PCI INTx) will need the polarity threaded + // through from discovery. Nothing danos binds today is such a device. + architecture.irqRoute(gsi, vector, true, false); + architecture.irqUnmask(gsi); +} + +/// Re-arm `gsi` after the driver has quieted the device. Caller holds the big lock +/// and has verified ownership. +pub fn ack(gsi: u32) bool { + if (gsi >= maximum_gsi or bound[gsi] == null) return false; + architecture.irqUnmask(gsi); + return true; +} + +/// Drop every binding made by task `owner` — called as that task exits, *before* its +/// endpoints are freed. Each line is left masked, so a dead driver's device goes quiet +/// rather than interrupting into a freed endpoint. Caller holds the big kernel lock, +/// which is what makes this safe against a concurrent `dispatch` on another core. +/// +/// Keyed on the owner, not the endpoint: endpoints are shared (a registered service's +/// endpoint has references in several processes), so releasing "everything pointing at +/// this endpoint" would tear down bindings this task never made. +pub fn releaseOwner(owner: u32) void { + for (&bound, 0..) |*slot, gsi| { + if (slot.* != null and bound_owner[gsi] == owner) { + architecture.irqMask(@intCast(gsi)); + slot.* = null; + bound_owner[gsi] = 0; + gsi_vector[gsi] = 0; + } + } +} diff --git a/src/kernel/log.zig b/src/kernel/log.zig index 7ca26f2..d7e51b4 100644 --- a/src/kernel/log.zig +++ b/src/kernel/log.zig @@ -19,18 +19,18 @@ //! and `recordPanic` (a breadcrumb in a fixed record). const std = @import("std"); -const arch = @import("arch"); +const architecture = @import("architecture"); pub const SinkFn = *const fn ([]const u8) void; -const max_sinks = 8; -var sinks: [max_sinks]SinkFn = undefined; +const maximum_sinks = 8; +var sinks: [maximum_sinks]SinkFn = undefined; var sink_count: usize = 0; /// Register an output sink. Every registered sink receives every message; sinks /// must be self-guarding (safe to call when their device is absent). pub fn addSink(sink: SinkFn) void { - if (sink_count < max_sinks) { + if (sink_count < maximum_sinks) { sinks[sink_count] = sink; sink_count += 1; } @@ -44,15 +44,15 @@ pub fn write(bytes: []const u8) void { /// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so /// this is safe to call from interrupt context and from a panic. pub fn print(comptime fmt: []const u8, args: anytype) void { - var buf: [256]u8 = undefined; - write(std.fmt.bufPrint(&buf, fmt, args) catch return); + var buffer: [256]u8 = undefined; + write(std.fmt.bufPrint(&buffer, fmt, args) catch return); } /// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available /// progress channel for when there is no text output at all. Independent of the /// sink list, so it works even before any sink is registered. pub fn checkpoint(code: u8) void { - arch.checkpoint(code); + architecture.checkpoint(code); } // --- persistent panic breadcrumb ------------------------------------------- @@ -68,16 +68,16 @@ pub const PanicRecord = extern struct { magic: u64 = 0, len: u32 = 0, _pad: u32 = 0, - msg: [512]u8 = undefined, + message: [512]u8 = undefined, }; /// Findable by symbol (`log.panic_record`) for a debugger or RAM dump. pub var panic_record: PanicRecord = .{}; /// Stamp the panic message into the breadcrumb record. -pub fn recordPanic(msg: []const u8) void { - const n: u32 = @intCast(@min(msg.len, panic_record.msg.len)); - @memcpy(panic_record.msg[0..n], msg[0..n]); +pub fn recordPanic(message: []const u8) void { + const n: u32 = @intCast(@min(message.len, panic_record.message.len)); + @memcpy(panic_record.message[0..n], message[0..n]); panic_record.len = n; panic_record.magic = panic_magic; // set last: a reader sees a complete record } diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 660227f..a07cb4a 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -1,24 +1,25 @@ const std = @import("std"); const danos = @import("danos"); -const config = @import("config"); -const arch = @import("arch"); +const parameters = @import("parameters"); +const architecture = @import("architecture"); const console = @import("console.zig"); const log = @import("log.zig"); const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const scheduler = @import("scheduler.zig"); const process = @import("process.zig"); -const devsvc = @import("devsvc.zig"); +const device_service = @import("device-service.zig"); +const irq = @import("irq.zig"); const initrd = @import("initrd"); const platform = @import("platform"); const tests = @import("tests.zig"); const build_options = @import("build_options"); -const BootInfo = danos.BootInfo; +const BootInformation = danos.BootInformation; -/// The calling convention used to enter the kernel. Pinned to SysV explicitly: +/// The calling convention used to enter the kernel. Pinned to SystemV explicitly: /// the bootloader is built for the UEFI target, whose C convention is Microsoft -/// x64 (first argument in RCX), while the kernel is SysV (first argument in -/// RDI). Both sides reference this so the `boot_info` pointer lands in the +/// x64 (first argument in RCX), while the kernel is SystemV (first argument in +/// RDI). Both sides reference this so the `boot_information` pointer lands in the /// register the other expects. `danos.kernel_abi` re-exports it to the loader. pub const kernel_abi = danos.kernel_abi; @@ -43,38 +44,38 @@ var ap_trampoline_page: u64 = 0; /// pointer to the handoff data. There is no runtime, no stack unwinding, and no /// caller to return to, so this never returns. /// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss -/// then calls this with the loader's `boot_info` pointer in RDI. We can't keep +/// then calls this with the loader's `boot_information` pointer in RDI. We can't keep /// running on the loader's stack: it's a low physical address that the identity /// map covers only transitionally, and vanishes once the kernel drops the low -/// half. `boot_info` (also low) is reached through the physmap — its base is the +/// half. `boot_information` (also low) is reached through the physmap — its base is the /// same under the loader's bootstrap tables and the kernel's own. -export fn kmainEntry(boot_info: *const BootInfo) callconv(kernel_abi) noreturn { - kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info)))); +export fn kmainEntry(boot_information: *const BootInformation) callconv(kernel_abi) noreturn { + kmain(@ptrFromInt(danos.physicalToVirtual(@intFromPtr(boot_information)))); } -fn kmain(boot_info: *const BootInfo) noreturn { +fn kmain(boot_information: *const BootInformation) noreturn { // The **log** is the machine-readable diagnostic stream: it fans out to every // *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a // file on a ramdisk/USB/SSD), so a message survives as long as any is present. // A headless, serial-less machine still boots correctly — it just goes quiet, // with port-0x80 checkpoints as the only progress signal. - arch.serialInit(); - log.addSink(arch.serialWrite); - if (arch.debugconPresent()) log.addSink(arch.debugconWrite); + architecture.serialInit(); + log.addSink(architecture.serialWrite); + if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite); // The **framebuffer** is deliberately *not* a log sink. It's a separate output // surface — a bootstrap text console today, a graphics device driver later — so // we never assume the OS is text-based. Only a few user-facing status lines // (via `status`) and panics are mirrored to it; the verbose log stays out. - const fb = boot_info.framebuffer; + const fb = boot_information.framebuffer; console.init(fb); log.checkpoint(cp_entry); // Catch CPU exceptions before doing anything that might fault: install our // reporter, then bring up the GDT + IDT. - arch.setFaultHandler(onException); - arch.init(); + architecture.setFaultHandler(onException); + architecture.init(); status("danos: initialising kernel...\n"); log.write(if (console.present()) @@ -90,7 +91,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Summarise the physical memory the loader handed us. The array is danos's // own MemoryRegion, so this is a plain slice — no firmware layout in sight. - const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(boot_info.memory_map.regions)))[0..boot_info.memory_map.len]; + const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len]; var usable_pages: u64 = 0; var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM for (regions) |r| { @@ -112,7 +113,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Bring up the physical frame allocator over that map, and prove it works: // allocate three frames, then hand them back. - pmm.init(boot_info.memory_map); + pmm.init(boot_information.memory_map); // Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap // draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held // until SMP bring-up; 0 means none was available (we stay uniprocessor). @@ -130,11 +131,11 @@ fn kmain(boot_info: *const BootInfo) noreturn { log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); // Switch off the firmware's page tables onto our own (with real permissions). - arch.enablePaging(pmm.alloc, pmm.free, boot_info); + architecture.enablePaging(pmm.alloc, pmm.free, boot_information); log.checkpoint(cp_paging); log.print("\ndanos: paging enabled\n", .{}); - log.print(" page tables: root = 0x{x:0>16}\n", .{arch.activePageTable()}); - log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count}); + log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()}); + log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count}); // Bring up the kernel heap (dynamic allocation), built on the VMM. heap.init(); @@ -146,24 +147,32 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Enumerate hardware from the firmware tables (ACPI here) into a generic // device tree, then list it. Discovery walks ACPI memory directly (identity- - // mapped) and maps PCIe config space on demand via the VMM. A failure here is + // mapped) and maps PCIe configuration space on demand via the VMM. A failure here is // not fatal yet — log it and carry on. const hal = platform.Hal{ - .mapMmio = arch.mapMmio, - .pioRead = arch.pioRead, - .pioWrite = arch.pioWrite, + .mapMmio = architecture.mapMmio, + .pioRead = architecture.pioRead, + .pioWrite = architecture.pioWrite, }; - if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| { - var dt = devtree; + if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| { + var device_tree = devtree; log.write("\ndanos: device discovery online\n"); - dt.dump(log.write); + device_tree.dump(log.write); - // Snapshot the device tree for user-space drivers (dev_enumerate/claim/ + // Snapshot the device tree for user-space drivers (device_enumerate/claim/ // mmio_map operate on this flat, id-indexed table + claim map). - devsvc.init(&dt); + device_service.init(&device_tree); + if (device_service.dropped > 0) { + // Otherwise entirely silent: drivers would just never see that hardware. + log.print("danos: WARNING {d} device(s) dropped — table full\n", .{device_service.dropped}); + } + + // Install the device-IRQ trampolines, so a driver's irq_bind has vectors to + // land on. Every line stays masked until something binds it (ioapic.init). + irq.init(); // Power register map extracted from the FADT + AML, for confidence it parsed. - const pw = platform.powerInfo(); + const pw = platform.powerInformation(); log.write("danos: power\n"); log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width }); if (pw.s5) |s| { @@ -177,32 +186,32 @@ fn kmain(boot_info: *const BootInfo) noreturn { const am = platform.amlStats(); log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total }); - // Feed the arch layer the discovered addresses/facts so it makes no legacy + // Feed the architecture layer the discovered addresses/facts so it makes no legacy // assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases // (HPET, I/O APIC) come from the device tree; scalar facts from ACPI. - const pinfo = platform.platformInfo(); - const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t| + const pinfo = platform.platformInformation(); + const hpet_base: u64 = if (device_tree.firstOfClass(.timer)) |t| (if (t.firstResource(.memory)) |r| r.start else 0) else 0; var ioapic_base: u64 = 0; var ioapic_gsi: u32 = 0; - if (dt.firstOfClass(.interrupt_controller)) |ic| { + if (device_tree.firstOfClass(.interrupt_controller)) |ic| { if (ic.firstResource(.memory)) |r| ioapic_base = r.start; if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start); } - var isos: [16]arch.IsoEntry = undefined; + var isos: [16]architecture.IsoEntry = undefined; const iso_n = @min(pinfo.override_count, isos.len); for (0..iso_n) |i| isos[i] = .{ .source = pinfo.overrides[i].source, .gsi = pinfo.overrides[i].gsi, .flags = pinfo.overrides[i].flags, }; - const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present()) + const pm_timer: ?architecture.PmTimer = if (pinfo.pm_timer.present()) .{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit } else null; - arch.configurePlatform(.{ + architecture.configurePlatform(.{ .pic_present = pinfo.pic_present, .hpet_base = hpet_base, .pm_timer = pm_timer, @@ -210,7 +219,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { .ioapic_gsi_base = ioapic_gsi, .overrides = isos[0..iso_n], }); - if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address); + if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address); log.write("danos: platform\n"); log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"}); @@ -222,7 +231,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { } else { log.write(" console UART: none in SPCR -> legacy COM1\n"); } - log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, arch.irqRouteCount(), arch.irqRouteRaw(0) }); + log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, architecture.irqRouteCount(), architecture.irqRouteRaw(0) }); const cores = platform.cpus(); log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 }); if (platform.cpusDropped() > 0) @@ -232,7 +241,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { } log.checkpoint(cp_discovery); - // Install the syscall handler (int 0x80 gate + syscall stub) once, before any + // Install the system_call handler (int 0x80 gate + system_call stub) once, before any // user code runs. process.init(); @@ -243,10 +252,10 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Start the timer and unmask interrupts — the kernel now has a heartbeat, and // the timer preempts among tasks. - arch.startTimer(); - arch.enableInterrupts(); + architecture.startTimer(); + architecture.enableInterrupts(); log.checkpoint(cp_timer); - log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.timerClockHz() / 1_000_000, arch.clockHz() / 1_000_000, arch.timerCalibrationSource() }); + log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() }); // Wake the other cores (application processors). A no-op on a single-core // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). @@ -255,8 +264,8 @@ fn kmain(boot_info: *const BootInfo) noreturn { // In a test build (`zig build -Dtest-case=`), run that case and stop. // Normal builds fall through to the idle halt. if (build_options.test_case) |case| { - tests.run(case, boot_info); - arch.halt(); + tests.run(case, boot_information); + architecture.halt(); } log.checkpoint(cp_running); @@ -266,9 +275,9 @@ fn kmain(boot_info: *const BootInfo) noreturn { // loader) and spawn it as a real ring-3 process, PID 1. It runs on its own // address space, preemptively, alongside the kernel — no cooperative // borrowing. This boot context then becomes the BSP's idle loop. - if (boot_info.init_len != 0) { + if (boot_information.init_len != 0) { status("starting /sbin/init...\n"); - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len]; process.spawnProcess(image, 4) catch |err| { statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)}); }; @@ -278,22 +287,22 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Spawn the extra user binaries the loader ferried in the initrd (the VFS // server, and later device drivers). For now the kernel launches them all; - // once init is a real service supervisor it will spawn them itself (sys_spawn). - startInitrdBinaries(boot_info); + // once init is a real service supervisor it will spawn them itself (system_spawn). + startInitrdBinaries(boot_information); // Become the idle task: drop below every real task and halt until an // interrupt. The timer keeps preempting into init and any other work. scheduler.setPriority(0); status("\nkernel idle; /sbin/init is running.\n"); - arch.halt(); + architecture.halt(); } /// Spawn every program bundled in the initrd as its own ring-3 process. A bad /// image or a program that fails to load is logged and skipped — the rest of the /// system still runs. -fn startInitrdBinaries(boot_info: *const danos.BootInfo) void { - if (boot_info.initrd_len == 0) return; - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.initrd_base)))[0..boot_info.initrd_len]; +fn startInitrdBinaries(boot_information: *const danos.BootInformation) void { + if (boot_information.initrd_len == 0) return; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len]; const rd = initrd.Reader.init(image) orelse { status("initrd: bad image, skipping\n"); return; @@ -323,40 +332,40 @@ fn bringUpSecondaries() void { log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n"); return; } - arch.setTrampolinePage(ap_trampoline_page); - arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop + architecture.setTrampolinePage(ap_trampoline_page); + architecture.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop // Test hook: the smp-retry case forces the first wake to fail, so the retry below // must still bring every core online. Inert in a normal build (test_case is null). if (build_options.test_case) |tc| { - if (std.mem.eql(u8, tc, "smp-retry")) arch.testFailNextWakes(1); + if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1); } log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1}); - const max_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried + const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried for (cores[1..], 1..) |core, index| { - const stack = heap.allocator().alloc(u8, config.kernel_stack_size) catch { + const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch { log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id}); continue; }; const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); // This core's dedicated fault stack — allocated only now that the core is // real, rather than reserved statically for every possible core. - const fault_stack = heap.allocator().alloc(u8, arch.fault_stack_size) catch { + const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch { log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id}); continue; }; - arch.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15)); + architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15)); const pc = scheduler.prepareSecondary(index, core.apic_id); var attempt: u32 = 1; - while (attempt <= max_wake_attempts) : (attempt += 1) { - if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { + while (attempt <= maximum_wake_attempts) : (attempt += 1) { + if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { pc.online = true; log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt }); break; } - if (attempt == max_wake_attempts) - log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, max_wake_attempts }); + if (attempt == maximum_wake_attempts) + log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts }); } } log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len }); @@ -365,14 +374,14 @@ fn bringUpSecondaries() void { /// A user-facing status line: to the diagnostic `log` *and* the on-screen console /// (if a framebuffer is present). The verbose log uses `log.*` directly and never /// touches the framebuffer. -fn status(msg: []const u8) void { - log.write(msg); - console.write(msg); +fn status(message: []const u8) void { + log.write(message); + console.write(message); } fn statusPrint(comptime fmt: []const u8, args: anytype) void { - var buf: [256]u8 = undefined; - status(std.fmt.bufPrint(&buf, fmt, args) catch return); + var buffer: [256]u8 = undefined; + status(std.fmt.bufPrint(&buffer, fmt, args) catch return); } /// Frames (4 KiB pages) to whole MiB. @@ -390,33 +399,33 @@ fn kib(frames: u64) u64 { /// running (full recovery — kill the task, keep the core — is the resilience track, /// see docs/resilience.md). The report names the core so an AP fault is attributed, /// and goes to every output sink plus a POST code and a persistent breadcrumb. -fn onException(state: *const arch.CpuState) noreturn { +fn onException(state: *const architecture.CpuState) noreturn { log.checkpoint(cp_exception); const core = scheduler.currentCpuIndex(); // A fault is user-facing enough to paint on screen too (via statusPrint), on // top of the diagnostic log. - statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.exceptionName(state.vector), state.vector }); + statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector }); statusPrint(" error code : 0x{x}\n", .{state.error_code}); - statusPrint(" IP : 0x{x:0>16}\n", .{arch.instructionPointer(state)}); - statusPrint(" SP : 0x{x:0>16}\n", .{arch.stackPointer(state)}); - if (arch.faultAddress(state)) |addr| statusPrint(" fault addr : 0x{x:0>16}\n", .{addr}); + statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)}); + statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)}); + if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address}); - var buf: [128]u8 = undefined; - log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ arch.exceptionName(state.vector), state.vector, core, arch.instructionPointer(state) }) catch "cpu exception"); - arch.halt(); + var buffer: [128]u8 = undefined; + log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception"); + architecture.halt(); } /// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a /// POST code + a persistent breadcrumb (so a post-mortem can recover it even with /// no live console), then halt. Assumes no console — the sinks self-guard. pub const panic = std.debug.FullPanic(struct { - fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn { - _ = first_trace_addr; + fn panic(message: []const u8, first_trace_address: ?usize) noreturn { + _ = first_trace_address; log.checkpoint(cp_panic); - log.recordPanic(msg); + log.recordPanic(message); status("\nKERNEL PANIC: "); - status(msg); + status(message); status("\n"); - arch.halt(); + architecture.halt(); } }.panic); diff --git a/src/kernel/pmm.zig b/src/kernel/pmm.zig index 7c62282..ec317e9 100644 --- a/src/kernel/pmm.zig +++ b/src/kernel/pmm.zig @@ -49,7 +49,7 @@ inline fn setFree(frame: usize) void { } fn regions(map: danos.MemoryMap) []const danos.MemoryRegion { - return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(map.regions)))[0..map.len]; + return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(map.regions)))[0..map.len]; } /// Build the allocator from the loader's memory map. Reaches physical memory @@ -91,7 +91,7 @@ pub fn init(map: danos.MemoryMap) void { } } const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap"); - bitmap = @as([*]u8, @ptrFromInt(danos.physToVirt(bitmap_base)))[0..bitmap_bytes]; + bitmap = @as([*]u8, @ptrFromInt(danos.physicalToVirtual(bitmap_base)))[0..bitmap_bytes]; // 3. Start with everything marked used, then free the usable regions. Doing // it this way means every gap, reserved span and MMIO hole is unallocatable @@ -165,8 +165,8 @@ pub fn allocBelow(limit: u64) ?u64 { /// Return a frame obtained from alloc() to the pool. Bogus or double frees are /// ignored rather than corrupting the count. -pub fn free(addr: u64) void { - const f: usize = @intCast(addr / page_size); +pub fn free(address: u64) void { + const f: usize = @intCast(address / page_size); if (f >= total_frames or !isUsed(f)) return; setFree(f); used_frames -= 1; diff --git a/src/kernel/process.zig b/src/kernel/process.zig index 3bd49c2..f5cede9 100644 --- a/src/kernel/process.zig +++ b/src/kernel/process.zig @@ -10,7 +10,7 @@ //! *current* kernel context via the borrowed-thread path — a minimal probe of //! the ring-transition mechanisms, kept for that test. //! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program -//! talks to the kernel only through the syscall instruction (or the int 0x80 +//! talks to the kernel only through the system_call instruction (or the int 0x80 //! gate). The shared handler is installed once by `init`. //! //! Borrowed-path caveat (`run` only): it publishes TSS.rsp0 on the *current* @@ -22,23 +22,24 @@ const std = @import("std"); const elf = std.elf; const danos = @import("danos"); -const arch = @import("arch"); +const architecture = @import("architecture"); const pmm = @import("pmm.zig"); -const sched = @import("scheduler.zig"); +const scheduler = @import("scheduler.zig"); const sync = @import("sync.zig"); -const ipc = @import("ipc_sync.zig"); -const devsvc = @import("devsvc.zig"); +const ipc = @import("ipc-synchronous.zig"); +const device_service = @import("device-service.zig"); +const irq = @import("irq.zig"); const log = @import("log.zig"); const page_size = danos.page_size; -const Syscall = danos.Syscall; +const SystemCall = danos.SystemCall; /// User virtual addresses. PML4 index 224 — a user-exclusive region, far from /// the identity map (low indices) and the vmm test address (index 128), so /// setting the U/S bit on its intermediate tables widens no kernel mapping. -/// An ELF image may occupy [code_virt, stack_virt); the stack page sits above. -pub const code_virt: u64 = 0x0000_7000_0000_0000; -pub const stack_virt: u64 = 0x0000_7000_0020_0000; +/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above. +pub const code_virtual: u64 = 0x0000_7000_0000_0000; +pub const stack_virtual: u64 = 0x0000_7000_0020_0000; /// The mmap grant arena: where `mmap` hands out fresh user pages, above the image /// and stack but still inside PML4[224] (so no kernel mapping is widened). Each @@ -48,19 +49,19 @@ pub const heap_arena_base: u64 = 0x0000_7000_1000_0000; pub const heap_arena_end: u64 = heap_arena_base + (1 << 30); /// End of the user (low) canonical half. Any legitimate user pointer is below it; -/// used to bound the addresses a syscall will dereference on the caller's behalf. +/// used to bound the addresses a system_call will dereference on the caller's behalf. pub const user_half_end: u64 = 0x0000_8000_0000_0000; /// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] — /// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping /// device pages user-accessible widens no kernel mapping. Per-process cursor in -/// `Task.dev_map_next`. -pub const dev_arena_base: u64 = 0x0000_7100_0000_0000; -pub const dev_arena_end: u64 = dev_arena_base + (4 << 30); +/// `Task.device_map_next`. +pub const device_arena_base: u64 = 0x0000_7100_0000_0000; +pub const device_arena_end: u64 = device_arena_base + (4 << 30); /// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small /// chunks, so this bound is generous; it also caps the frame scratch array below. -const max_mmap_pages = 256; +const maximum_mmap_pages = 256; // The hand-assembled user program blob (isr.s, .rodata) — the isolation probe. const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" }); @@ -71,167 +72,251 @@ pub fn pfBlob() []const u8 { return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)]; } -/// What debug_write syscalls produced (accumulated), and the exit syscall's code. -pub var write_buf: [256]u8 = undefined; +/// What debug_write syscalls produced (accumulated), and the exit system_call's code. +pub var write_buffer: [256]u8 = undefined; pub var write_len: usize = 0; pub var write_from_user: bool = false; pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests) pub var exit_code: u64 = 0; -/// The syscall surface, dispatched on the saved syscall number (`danos.Syscall`). +/// The system_call surface, dispatched on the saved system_call number (`danos.SystemCall`). /// This is the microkernel-minimal set — memory + scheduling only; file/device /// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is /// written back into the trap frame, since the entry paths restore user registers -/// from it. One handler serves both the syscall/sysret and int-0x80 entry paths. +/// from it. One handler serves both the system_call/sysret and int-0x80 entry paths. /// /// Install it once at boot (before any user code runs) via `init`. pub fn init() void { - arch.setSyscallHandler(syscall); + architecture.setSystemCallHandler(system_call); } -/// Return -1 (as an unsigned bit pattern) in the syscall result register. -fn fail(state: *arch.CpuState) void { - arch.setSyscallResult(state, @bitCast(@as(i64, -1))); +/// Return -1 (as an unsigned bit pattern) in the system_call result register. +fn fail(state: *architecture.CpuState) void { + architecture.setSystemCallResult(state, @bitCast(@as(i64, -1))); } -fn syscall(state: *arch.CpuState) void { - switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) { +fn system_call(state: *architecture.CpuState) void { + switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) { .exit => { - exit_code = arch.syscallArg(state, 0); + exit_code = architecture.systemCallArg(state, 0); // A scheduled process drops its endpoint references, frees its address // space, and reschedules; a borrowed test thread unwinds back to the // kernel that entered it. - if (sched.currentIsUserProcess()) { - ipc.closeHandles(sched.cur()); - sched.exitUser(); - } else arch.userExit(); + if (scheduler.currentIsUserProcess()) { + // Unbind before closeHandles: dropping the last reference destroys the + // Endpoint, and a still-bound GSI would have an ISR call + // notifyFromIsr on freed memory the next time the device fired. + // unbindAll also leaves the line masked, so a dead driver's device + // goes quiet rather than storming. + releaseIrqs(scheduler.current()); + ipc.closeHandles(scheduler.current()); + scheduler.exitUser(); + } else architecture.userExit(); }, .yield => { - sched.yield(); - arch.setSyscallResult(state, 0); + scheduler.yield(); + architecture.setSystemCallResult(state, 0); }, .sleep => { - sched.sleep(arch.syscallArg(state, 0)); - arch.setSyscallResult(state, 0); + scheduler.sleep(architecture.systemCallArg(state, 0)); + architecture.setSystemCallResult(state, 0); }, - .debug_write => sysDebugWrite(state), - .mmap => sysMmap(state), - .munmap => sysMunmap(state), - .create_endpoint => sysCreateEndpoint(state), - .ipc_register => sysIpcRegister(state), - .ipc_lookup => sysIpcLookup(state), - .ipc_call => sysIpcCall(state), - .ipc_reply_wait => sysIpcReplyWait(state), - .dev_enumerate => sysDevEnumerate(state), - .dev_claim => sysDevClaim(state), - .mmio_map => sysMmioMap(state), - // irq_bind/irq_ack (IRQ-as-message) land with the driver that needs them. - .irq_bind, .irq_ack => fail(state), + .debug_write => systemDebugWrite(state), + .mmap => systemMmap(state), + .munmap => systemMunmap(state), + .create_endpoint => systemCreateEndpoint(state), + .ipc_register => systemIpcRegister(state), + .ipc_lookup => systemIpcLookup(state), + .ipc_call => systemIpcCall(state), + .ipc_reply_wait => systemIpcReplyWait(state), + .device_enumerate => systemDeviceEnumerate(state), + .device_claim => systemDeviceClaim(state), + .mmio_map => systemMmioMap(state), + .irq_bind => systemIrqBind(state), + .irq_ack => systemIrqAck(state), + .device_register => systemDeviceRegister(state), _ => fail(state), } } -/// Return `-errno` in the syscall result register. -fn failErr(state: *arch.CpuState, errno: i64) void { - arch.setSyscallResult(state, @bitCast(-errno)); +/// Return `-errno` in the system_call result register. +fn failErr(state: *architecture.CpuState, errno: i64) void { + architecture.setSystemCallResult(state, @bitCast(-errno)); } /// create_endpoint() -> handle: allocate an endpoint and install it in the /// caller's handle table. -fn sysCreateEndpoint(state: *arch.CpuState) void { - const ep = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM); - const h = ipc.installHandle(sched.cur(), ep); +fn systemCreateEndpoint(state: *architecture.CpuState) void { + const endpoint = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM); + const h = ipc.installHandle(scheduler.current(), endpoint); if (h < 0) { - ipc.dropRef(ep); + ipc.dropRef(endpoint); return failErr(state, ipc.ENOSPC); } - arch.setSyscallResult(state, @intCast(h)); + architecture.setSystemCallResult(state, @intCast(h)); } /// ipc_register(service_id, handle): publish the caller's endpoint under a /// well-known id so other processes can find it. -fn sysIpcRegister(state: *arch.CpuState) void { - const id: u32 = @truncate(arch.syscallArg(state, 0)); - const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 1)) orelse return failErr(state, ipc.EBADF); - arch.setSyscallResult(state, @bitCast(ipc.register(id, ep))); +fn systemIpcRegister(state: *architecture.CpuState) void { + const id: u32 = @truncate(architecture.systemCallArg(state, 0)); + const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF); + architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint))); } /// ipc_lookup(service_id) -> handle: find a published endpoint and install a /// handle to it in the caller. -fn sysIpcLookup(state: *arch.CpuState) void { - const id: u32 = @truncate(arch.syscallArg(state, 0)); - const ep = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT); - const h = ipc.installHandle(sched.cur(), ep); +fn systemIpcLookup(state: *architecture.CpuState) void { + const id: u32 = @truncate(architecture.systemCallArg(state, 0)); + const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT); + const h = ipc.installHandle(scheduler.current(), endpoint); if (h < 0) { - ipc.dropRef(ep); + ipc.dropRef(endpoint); return failErr(state, ipc.ENOSPC); } - arch.setSyscallResult(state, @intCast(h)); + architecture.setSystemCallResult(state, @intCast(h)); } -/// ipc_call(handle, msg_ptr, msg_len, reply_ptr, reply_cap) -> reply_len. +/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len. /// Blocks until the server replies; the trap frame lives on this task's kernel /// stack, so it survives the block and receives the result on resume. -fn sysIpcCall(state: *arch.CpuState) void { - const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF); - const r = ipc.call(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4)); - arch.setSyscallResult(state, @bitCast(r)); +fn systemIpcCall(state: *architecture.CpuState) void { + const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); + const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4)); + architecture.setSystemCallResult(state, @bitCast(r)); } -/// ipc_reply_wait(handle, reply_ptr, reply_len, recv_ptr, recv_cap) -> recv_len, +/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len, /// with the sender's badge in the secondary result register (rdx). -fn sysIpcReplyWait(state: *arch.CpuState) void { - const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF); +fn systemIpcReplyWait(state: *architecture.CpuState) void { + const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); var badge: u64 = 0; - const r = ipc.replyWait(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4), &badge); - arch.setSyscallResult(state, @bitCast(r)); - arch.setSyscallResult2(state, badge); + const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge); + architecture.setSystemCallResult(state, @bitCast(r)); + architecture.setSystemCallResult2(state, badge); } -/// dev_enumerate(buf, max) -> total: snapshot the device table into the caller's -/// buffer (up to `max` entries), returning the total device count. -fn sysDevEnumerate(state: *arch.CpuState) void { - const buf_ptr = arch.syscallArg(state, 0); - const max = arch.syscallArg(state, 1); - const t = sched.cur(); - if (t.aspace == 0 or buf_ptr >= user_half_end) return fail(state); - const sz = @sizeOf(danos.DeviceDesc); - const cap = @min(max, (user_half_end - buf_ptr) / sz); // clamp to the user half - const out: [*]danos.DeviceDesc = @ptrFromInt(buf_ptr); - arch.setSyscallResult(state, devsvc.enumerate(out[0..@intCast(cap)])); +/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's +/// buffer (up to `maximum` entries), returning the total device count. +fn systemDeviceEnumerate(state: *architecture.CpuState) void { + const buffer_ptr = architecture.systemCallArg(state, 0); + const maximum = architecture.systemCallArg(state, 1); + const t = scheduler.current(); + if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state); + const sz = @sizeOf(danos.DeviceDescriptor); + const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half + const out: [*]danos.DeviceDescriptor = @ptrFromInt(buffer_ptr); + architecture.setSystemCallResult(state, device_service.enumerate(out[0..@intCast(cap)])); } -/// dev_claim(id) -> 0/-1: take exclusive ownership of a device for this process. -fn sysDevClaim(state: *arch.CpuState) void { - if (devsvc.claim(arch.syscallArg(state, 0), sched.cur().id)) - arch.setSyscallResult(state, 0) +/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process. +fn systemDeviceClaim(state: *architecture.CpuState) void { + if (device_service.claim(architecture.systemCallArg(state, 0), scheduler.current().id)) + architecture.setSystemCallResult(state, 0) else fail(state); } -/// mmio_map(dev_id, res_idx) -> vaddr: map a claimed device's MMIO window into +/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into /// this address space (strong-uncacheable) and return the register base address. /// The claim is the capability — a process can only map hardware it owns. -fn sysMmioMap(state: *arch.CpuState) void { - const dev_id = arch.syscallArg(state, 0); - const res_idx = arch.syscallArg(state, 1); - const t = sched.cur(); +fn systemMmioMap(state: *architecture.CpuState) void { + const device_id = architecture.systemCallArg(state, 0); + const resource_index = architecture.systemCallArg(state, 1); + const t = scheduler.current(); if (t.aspace == 0) return fail(state); - const owner = devsvc.ownerOf(dev_id) orelse return fail(state); + const owner = device_service.ownerOf(device_id) orelse return fail(state); if (owner != t.id) return fail(state); // not claimed by this process - const r = devsvc.resourceOf(dev_id, res_idx) orelse return fail(state); + const r = device_service.resourceOf(device_id, resource_index) orelse return fail(state); if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state); - if (t.dev_map_next == 0) t.dev_map_next = dev_arena_base; + if (t.device_map_next == 0) t.device_map_next = device_arena_base; const first = r.start & ~@as(u64, page_size - 1); const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1); const pages = (last - first) / page_size + 1; - const base_v = t.dev_map_next; - if (base_v + pages * page_size > dev_arena_end) return fail(state); + const base_v = t.device_map_next; + if (base_v + pages * page_size > device_arena_end) return fail(state); - arch.mapUserDeviceInto(t.aspace, base_v, r.start, r.len); - t.dev_map_next = base_v + pages * page_size; - arch.setSyscallResult(state, base_v + (r.start & (page_size - 1))); // register base + architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len); + t.device_map_next = base_v + pages * page_size; + architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base +} + +/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device +/// this process has claimed. The bus-driver primitive: a process that owns a bus +/// enumerates it and hands each device it finds to the table, where a class driver +/// can claim it. +/// +/// The kernel copies the descriptor into a kernel local *once* (via the same +/// physmap-walking path as IPC, so an unmapped user page fails the call rather than +/// faulting the kernel), then validates and uses that copy — no second read of user +/// memory, so nothing it checked can change under it. It refuses any child resource +/// that escapes the parent's windows: a descriptor is a licence to map physical +/// memory, so a bus may only subdivide what it already holds. `id`/`parent` in the +/// supplied descriptor are ignored. +fn systemDeviceRegister(state: *architecture.CpuState) void { + const parent_id = architecture.systemCallArg(state, 0); + const descriptor_ptr = architecture.systemCallArg(state, 1); + const t = scheduler.current(); + if (t.aspace == 0) return fail(state); + + var descriptor: danos.DeviceDescriptor = undefined; + if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state); + + const id = device_service.register(parent_id, t.id, &descriptor) catch return fail(state); + architecture.setSystemCallResult(state, id); +} + +/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed +/// (which is what frees the endpoints an ISR would otherwise notify into). +fn releaseIrqs(t: *scheduler.Task) void { + const flags = sync.enter(); + defer sync.leave(flags); + irq.releaseOwner(t.id); +} + +/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null. +/// The two checks are the whole security story: the device must be *claimed* by the +/// caller, and the resource must be one of that device's `irq` resources as recorded +/// by discovery. Neither a raw GSI nor an unclaimed device can get through — which +/// is why irq_bind takes a resource index and not an interrupt number. +fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 { + const owner = device_service.ownerOf(device_id) orelse return null; + if (owner != t.id) return null; + const r = device_service.resourceOf(device_id, resource_index) orelse return null; + if (r.kind != @intFromEnum(danos.ResourceKind.irq)) return null; + if (r.start >= irq.maximum_gsi) return null; + return @intCast(r.start); +} + +/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the +/// endpoint as an asynchronous IPC notification. The driver then blocks in +/// IPC_ReplyWait and is woken by the ISR; see src/kernel/irq.zig for the cycle. +fn systemIrqBind(state: *architecture.CpuState) void { + const t = scheduler.current(); + if (t.aspace == 0) return fail(state); + const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse + return fail(state); + const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state); + + const flags = sync.enter(); + defer sync.leave(flags); + irq.bind(gsi, endpoint, t.id) catch return fail(state); + architecture.setSystemCallResult(state, 0); +} + +/// irq_ack(device_id, resource_index) -> 0/-1: re-arm a bound IRQ. The ISR left the line +/// masked (it could not quiet the device — that's this driver's job), so nothing +/// more arrives until the driver says it has serviced the hardware. +fn systemIrqAck(state: *architecture.CpuState) void { + const t = scheduler.current(); + if (t.aspace == 0) return fail(state); + const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse + return fail(state); + + const flags = sync.enter(); + defer sync.leave(flags); + if (irq.ack(gsi)) architecture.setSystemCallResult(state, 0) else fail(state); } /// debug_write(ptr, len): copy bytes from user memory into the kernel log. @@ -243,18 +328,18 @@ fn sysMmioMap(state: *arch.CpuState) void { /// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in /// the user half passes the check and the read #PFs -> on_fault halts — a /// self-DoS, not an isolation break. Fault-recovering copy-in is a later item. -fn sysDebugWrite(state: *arch.CpuState) void { - const ptr = arch.syscallArg(state, 0); - const len = arch.syscallArg(state, 1); - if (len <= write_buf.len and ptr < user_half_end and ptr + len <= user_half_end) { - const src: [*]const u8 = @ptrFromInt(ptr); - @memcpy(write_buf[0..len], src[0..len]); // keep the latest message +fn systemDebugWrite(state: *architecture.CpuState) void { + const ptr = architecture.systemCallArg(state, 0); + const len = architecture.systemCallArg(state, 1); + if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) { + const source: [*]const u8 = @ptrFromInt(ptr); + @memcpy(write_buffer[0..len], source[0..len]); // keep the latest message write_len = len; - write_from_user = arch.fromUser(state); + write_from_user = architecture.fromUser(state); write_count += 1; log.write("DANOS-INIT: "); - log.write(src[0..len]); - arch.setSyscallResult(state, len); + log.write(source[0..len]); + architecture.setSystemCallResult(state, len); } else { fail(state); } @@ -264,13 +349,13 @@ fn sysDebugWrite(state: *arch.CpuState) void { /// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the /// base virtual address. `prot` is accepted but not yet honoured (grants are /// always RW+NX; W^X for user code stays with the ELF loader). Failure returns -/// -1. The user-space allocator (lib `rt`) carves these pages into malloc blocks. -fn sysMmap(state: *arch.CpuState) void { - const len = arch.syscallArg(state, 0); - const t = sched.cur(); +/// -1. The user-space allocator (lib `runtime`) carves these pages into malloc blocks. +fn systemMmap(state: *architecture.CpuState) void { + const len = architecture.systemCallArg(state, 0); + const t = scheduler.current(); if (t.aspace == 0) return fail(state); // not a user process — nothing to map into const pages = (len + page_size - 1) / page_size; - if (pages == 0 or pages > max_mmap_pages) return fail(state); + if (pages == 0 or pages > maximum_mmap_pages) return fail(state); if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily const base = t.heap_next; @@ -278,7 +363,7 @@ fn sysMmap(state: *arch.CpuState) void { // Reserve all frames up front so a mid-way exhaustion rolls back cleanly // (no partially-mapped grant leaks into the address space). - var frames: [max_mmap_pages]u64 = undefined; + var frames: [maximum_mmap_pages]u64 = undefined; var got: usize = 0; while (got < pages) : (got += 1) { frames[got] = pmm.alloc() orelse { @@ -288,12 +373,12 @@ fn sysMmap(state: *arch.CpuState) void { } for (frames[0..pages], 0..) |frame, i| { - const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame)); - @memset(dst[0..page_size], 0); // hand out zeroed memory - arch.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX + const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame)); + @memset(destination[0..page_size], 0); // hand out zeroed memory + architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX } t.heap_next = base + pages * page_size; - arch.setSyscallResult(state, base); + architecture.setSystemCallResult(state, base); } /// munmap(base, len): release a range previously handed out by `mmap`. Unmaps @@ -301,25 +386,25 @@ fn sysMmap(state: *arch.CpuState) void { /// range is not recycled (the user-space allocator reuses freed *blocks* itself); /// this just returns the physical frames to the kernel. Returns 0, or -1 if the /// range is not page-aligned or lies outside the arena. -fn sysMunmap(state: *arch.CpuState) void { - const base = arch.syscallArg(state, 0); - const len = arch.syscallArg(state, 1); - const t = sched.cur(); +fn systemMunmap(state: *architecture.CpuState) void { + const base = architecture.systemCallArg(state, 0); + const len = architecture.systemCallArg(state, 1); + const t = scheduler.current(); if (t.aspace == 0 or base % page_size != 0) return fail(state); const pages = (len + page_size - 1) / page_size; if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state); for (0..pages) |i| { const va = base + i * page_size; - if (arch.translate(t.aspace, va)) |phys| { - arch.unmapUserPageInto(t.aspace, va); - pmm.free(phys); + if (architecture.translate(t.aspace, va)) |physical| { + architecture.unmapUserPageInto(t.aspace, va); + pmm.free(physical); } } - arch.setSyscallResult(state, 0); + architecture.setSystemCallResult(state, 0); } -/// Reset the recorded syscall evidence before a user-mode run. +/// Reset the recorded system_call evidence before a user-mode run. fn resetRecords() void { write_len = 0; write_from_user = false; @@ -329,8 +414,8 @@ fn resetRecords() void { pub const RunError = error{ ProgramTooBig, OutOfMemory }; -/// Map `blob` at code_virt with a fresh user stack, drop to ring 3, and return -/// once the program exits via syscall 0. See the migration caveat in the module +/// Map `blob` at code_virtual with a fresh user stack, drop to ring 3, and return +/// once the program exits via system_call 0. See the migration caveat in the module /// doc. A program that faults instead never returns (on_fault halts the core). pub fn run(blob: []const u8) RunError!void { if (blob.len > page_size) return error.ProgramTooBig; @@ -343,20 +428,20 @@ pub fn run(blob: []const u8) RunError!void { // Fill the code frame through the physmap (supervisor RW): the user-facing // mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is // padded with int3 so a stray jump traps instead of sliding. - const code: [*]u8 = @ptrFromInt(danos.physToVirt(code_frame)); + const code: [*]u8 = @ptrFromInt(danos.physicalToVirtual(code_frame)); @memcpy(code[0..blob.len], blob); @memset(code[blob.len..page_size], 0xCC); - arch.mapUserPage(code_virt, code_frame, false, true); // RO + X - arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX + architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X + architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX resetRecords(); - arch.enterUser(sched.currentCpuIndex(), code_virt, stack_virt + page_size); + architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size); - // Back via the exit syscall; the interrupt gate left IF clear. - arch.enableInterrupts(); - arch.unmapPage(code_virt); - arch.unmapPage(stack_virt); + // Back via the exit system_call; the interrupt gate left IF clear. + architecture.enableInterrupts(); + architecture.unmapPage(code_virtual); + architecture.unmapPage(stack_virtual); pmm.free(code_frame); pmm.free(stack_frame); } @@ -371,8 +456,8 @@ pub const InitError = error{ OutOfMemory, }; -const max_segments = 16; -const max_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway +const maximum_segments = 16; +const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway const Segment = struct { vaddr: u64, @@ -391,7 +476,7 @@ const Segment = struct { /// against the image and the user region; segments must be page-aligned, /// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only /// segment, mapped RO+NX). -fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!struct { count: usize, entry: u64 } { +fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!struct { count: usize, entry: u64 } { if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf; const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]); if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf; @@ -399,7 +484,7 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru if (ehdr.e_machine != .X86_64) return error.BadElf; if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf; - if (ehdr.e_phnum > max_segments) return error.BadElf; + if (ehdr.e_phnum > maximum_segments) return error.BadElf; const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize; if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf; @@ -415,8 +500,8 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment; if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment; // Inside the user image region, strictly below the stack page. - if (phdr.p_vaddr < code_virt) return error.BadSegment; - if (phdr.p_memsz > stack_virt - phdr.p_vaddr) return error.BadSegment; + if (phdr.p_vaddr < code_virtual) return error.BadSegment; + if (phdr.p_memsz > stack_virtual - phdr.p_vaddr) return error.BadSegment; const w = phdr.p_flags & elf.PF_W != 0; const x = phdr.p_flags & elf.PF_X != 0; @@ -437,7 +522,7 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment; } total_pages += seg.pages(); - if (total_pages > max_pages) return error.ProgramTooBig; + if (total_pages > maximum_pages) return error.ProgramTooBig; segs[count] = seg; count += 1; } @@ -457,38 +542,38 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru /// frame mapped into it — so no per-page rollback list is needed here. fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void { const frame = pmm.alloc() orelse return error.OutOfMemory; - const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame)); - @memset(dst[0..page_size], 0); + const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame)); + @memset(destination[0..page_size], 0); const page_off = page_index * page_size; if (page_off < seg.filesz) { const n = @min(page_size, seg.filesz - page_off); - @memcpy(dst[0..n], image[seg.off + page_off ..][0..n]); + @memcpy(destination[0..n], image[seg.off + page_off ..][0..n]); } - arch.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable); + architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable); } /// Load a user ELF image into a fresh address space and spawn it as a scheduled /// ring-3 process at `priority`. Returns immediately — the process runs /// preemptively on its own page tables alongside everything else, and its exit -/// is handled by the syscall layer. The whole build (address space + ELF load + +/// is handled by the system_call layer. The whole build (address space + ELF load + /// task) runs under the kernel lock so it appears atomically and can't race /// pmm/heap on another core. pub fn spawnProcess(image: []const u8, priority: u3) InitError!void { - var segs: [max_segments]Segment = undefined; + var segs: [maximum_segments]Segment = undefined; const parsed = try parseSegments(image, &segs); const flags = sync.enter(); defer sync.leave(flags); - const aspace = arch.createAddressSpace() orelse return error.OutOfMemory; - errdefer arch.destroyAddressSpace(aspace); + const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory; + errdefer architecture.destroyAddressSpace(aspace); for (segs[0..parsed.count]) |seg| { for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i); } const stack_frame = pmm.alloc() orelse return error.OutOfMemory; - arch.mapUserPageInto(aspace, stack_virt, stack_frame, true, false); // RW + NX + architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX - if (!sched.spawnUserLocked(aspace, parsed.entry, stack_virt + page_size, priority)) + if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority)) return error.OutOfMemory; } diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 5236d17..2705908 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -18,17 +18,17 @@ //! shared queues. const std = @import("std"); -const config = @import("config"); -const arch = @import("arch"); +const parameters = @import("parameters"); +const architecture = @import("architecture"); const heap = @import("heap.zig"); const sync = @import("sync.zig"); /// Priority level: 0 (lowest) .. 7 (highest). 8 levels total. pub const Priority = u3; -const num_priorities = 8; +const number_priorities = 8; -const stack_size = config.kernel_stack_size; // each task's kernel stack -const max_tasks = config.max_tasks; // maximum tasks alive at once (static pool) +const stack_size = parameters.kernel_stack_size; // each task's kernel stack +const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool) const State = enum { free, ready, running, blocked }; @@ -52,11 +52,11 @@ pub const Task = struct { heap_next: u64 = 0, // Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 = // uninitialised, process.zig seeds it on the first mmio_map). User task only. - dev_map_next: u64 = 0, + device_map_next: u64 = 0, // --- synchronous IPC (ipc_sync.zig) --- // Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept // opaque here so the scheduler and IPC modules don't import each other. - handles: [ipc_max_handles]?*anyopaque = .{null} ** ipc_max_handles, + handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles, // A server holds the caller it currently owes a reply to (set by ReplyWait's // receive, cleared when it replies). A client, while blocked in Call, records // its message + reply buffers here and its result lands in `ipc_status`. @@ -71,13 +71,13 @@ pub const Task = struct { /// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because /// it dimensions a field of `Task`; ipc_sync.zig re-exports it. -pub const ipc_max_handles = 16; +pub const ipc_maximum_handles = 16; -var tasks = [_]Task{.{}} ** max_tasks; +var tasks = [_]Task{.{}} ** maximum_tasks; var next_id: u32 = 1; /// Per-CPU scheduler state: the task each core is running, its own idle task, and a -/// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a +/// queue of tasks **pinned** to it. One entry per core; the architecture layer stashes a /// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it /// with a single read and no lock. /// @@ -95,30 +95,30 @@ pub const PerCpu = struct { online: bool = false, // has this core finished bring-up? loaded_aspace: u64 = 0, // the address-space root currently loaded on this core // Tasks pinned to this core (affinity == index), per priority level + bitmap. - pinned_head: [num_priorities]?*Task = .{null} ** num_priorities, - pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities, + pinned_head: [number_priorities]?*Task = .{null} ** number_priorities, + pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities, pinned_bitmap: u8 = 0, }; -const max_cpus = config.max_cpus; -var cpus = [_]PerCpu{.{}} ** max_cpus; +const maximum_cpus = parameters.maximum_cpus; +var cpus = [_]PerCpu{.{}} ** maximum_cpus; -/// This core's per-CPU state, via the arch layer's GS-base pointer. Valid only once +/// This core's per-CPU state, via the architecture layer's GS-base pointer. Valid only once /// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`). inline fn thisCpu() *PerCpu { - return @ptrFromInt(arch.cpuLocal()); + return @ptrFromInt(architecture.cpuLocal()); } /// The task running on this core — the per-CPU replacement for the old global /// `current`. A convenience reader; writes go through `thisCpu().current`. -pub inline fn cur() *Task { +pub inline fn current() *Task { return thisCpu().current; } // Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These // are shared across all cores and mutated only under the big kernel lock. -var ready_head: [num_priorities]?*Task = .{null} ** num_priorities; -var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities; +var ready_head: [number_priorities]?*Task = .{null} ** number_priorities; +var ready_tail: [number_priorities]?*Task = .{null} ** number_priorities; var ready_bitmap: u8 = 0; var preemption_enabled = true; @@ -129,12 +129,12 @@ var preemption_enabled = true; /// boot, before interrupts are enabled — so no lock is needed here. pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; - pc.* = .{ .index = 0, .online = true, .loaded_aspace = arch.kernelPageTable() }; - arch.setCpuLocal(0, @intFromPtr(pc)); + pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() }; + architecture.setCpuLocal(0, @intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority - arch.setTickHook(tick); + architecture.setTickHook(tick); } /// The idle task: run when every other task is blocked or sleeping. `hlt` waits @@ -145,7 +145,7 @@ fn idle() void { /// Reserve and initialise the per-CPU slot for an application processor at dense /// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a -/// pointer the arch bring-up hands to the core (it publishes it in its GS base). +/// pointer the architecture bring-up hands to the core (it publishes it in its GS base). /// Called on the BSP before waking each AP; the AP marks itself `online`. pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu { const pc = &cpus[index]; @@ -153,12 +153,12 @@ pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu { return pc; } -/// Entry for an application processor once the arch layer has set up its per-CPU +/// Entry for an application processor once the architecture layer has set up its per-CPU /// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task /// (as task 0 is for the BSP), marks the core online, and enters the run loop: with /// interrupts enabled the timer preempts this idle context into whatever the global /// ready queue offers, so the core runs real work in parallel with the others. The -/// `.c` calling convention lets the arch trampoline path jump here. Never returns. +/// `.c` calling convention lets the architecture trampoline path jump here. Never returns. pub fn secondaryMain() callconv(.c) noreturn { const flags = sync.enter(); const pc = thisCpu(); @@ -168,10 +168,10 @@ pub fn secondaryMain() callconv(.c) noreturn { pc.current = t; pc.idle = t; pc.online = true; - pc.loaded_aspace = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up + pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up sync.leave(flags); - arch.enableInterrupts(); // the timer now preempts this idle context into work + architecture.enableInterrupts(); // the timer now preempts this idle context into work while (true) asm volatile ("hlt"); // idle when this core has nothing ready } @@ -195,7 +195,7 @@ fn enqueue(t: *Task) void { } } -fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void { +fn enqueueTo(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void { t.next = null; const p: usize = t.priority; if (tail[p]) |tl| tl.next = t else head[p] = t; @@ -206,7 +206,7 @@ fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitma /// The highest non-empty priority level in a bitmap, or -1 if empty. fn topLevel(bitmap: u8) i32 { if (bitmap == 0) return -1; - return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap)); + return @as(i32, number_priorities - 1) - @as(i32, @clz(bitmap)); } /// Pick the highest-priority ready task for core `pc`: the better of the global queue @@ -220,7 +220,7 @@ fn dequeueHighest(pc: *PerCpu) ?*Task { return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g)); } -fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task { +fn dequeueFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, level: usize) ?*Task { const t = head[level].?; head[level] = t.next; if (head[level] == null) { @@ -249,7 +249,7 @@ pub fn spawn(entry: *const fn () void, priority: Priority) void { pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { const flags = sync.enter(); defer sync.leave(flags); - const ok = cpu < max_cpus and cpus[cpu].online; + const ok = cpu < maximum_cpus and cpus[cpu].online; _ = create(entry, priority, if (ok) cpu else null); return ok; } @@ -277,7 +277,7 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority t.kstack_top = top; // First switch-in lands in startUserTask (no register smuggling — it reads // the user entry/stack from the Task itself). - t.sp = arch.initTaskStack(top, @intFromPtr(&startUserTask)); + t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask)); enqueue(t); return true; } @@ -287,10 +287,10 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority /// Task avoids smuggling values through callee-saved registers across the /// context switch and lock release. fn startUserTask() void { - const t = cur(); - var buf: [96]u8 = undefined; - arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch ""); - arch.jumpToUser(t.user_ip, t.user_sp); // noreturn + const t = current(); + var buffer: [96]u8 = undefined; + architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch ""); + architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn } /// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the @@ -303,7 +303,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task { next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; t.kstack_top = top; - t.sp = arch.initTaskStack(top, @intFromPtr(entry)); + t.sp = architecture.initTaskStack(top, @intFromPtr(entry)); enqueue(t); return t; } @@ -318,22 +318,22 @@ fn freeSlot() ?*Task { /// Pick the highest-priority ready task and switch this core to it. The big kernel /// lock must be held by the caller (which also keeps local interrupts disabled); /// it serialises every core's scheduling, so no other core can touch the shared -/// queues while we requeue `prev` and dequeue `next`. A dequeued task is `.ready`, +/// queues while we requeue `previous` and dequeue `next`. A dequeued task is `.ready`, /// never running elsewhere, so two cores never run the same task. fn schedule() void { const pc = thisCpu(); - const prev = pc.current; - if (prev.state == .running) { - prev.state = .ready; - enqueue(prev); // back of its level's queue (round-robin) + const previous = pc.current; + if (previous.state == .running) { + previous.state = .ready; + enqueue(previous); // back of its level's queue (round-robin) } const next = dequeueHighest(pc) orelse { - prev.state = .running; // nothing else ready — keep running + previous.state = .running; // nothing else ready — keep running return; }; next.state = .running; pc.current = next; - if (next != prev) switchTo(pc, &prev.sp, next); + if (next != previous) switchTo(pc, &previous.sp, next); } /// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a @@ -346,13 +346,13 @@ fn schedule() void { /// interrupt can observe a half-updated (kernel stack, address space) pair. /// `save_sp` receives the outgoing task's stack pointer. fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void { - if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top); - const want = if (next.aspace != 0) next.aspace else arch.kernelPageTable(); + if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top); + const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable(); if (want != pc.loaded_aspace) { - arch.loadPageTable(want); + architecture.loadPageTable(want); pc.loaded_aspace = want; } - arch.switchContext(save_sp, next.sp); + architecture.switchContext(save_sp, next.sp); } /// Voluntarily give up the CPU to the next ready task. @@ -366,8 +366,8 @@ pub fn yield() void { /// again. The idle task (or other work) runs in the meantime. pub fn sleep(ms: u64) void { const flags = sync.enter(); - const t = cur(); - t.wake_at = arch.millis() + ms; + const t = current(); + t.wake_at = architecture.millis() + ms; t.state = .blocked; schedule(); // current is blocked, so schedule() won't re-enqueue it sync.leave(flags); @@ -384,37 +384,37 @@ pub const WaitQueue = struct { head: ?*Task = null, }; -/// Block the current task on `wq` and switch away. Precondition: the big kernel +/// Block the current task on `wait_queue` and switch away. Precondition: the big kernel /// lock is held (so a condition can be checked and the block committed atomically; /// it also keeps local interrupts disabled). On return — when woken — the lock is /// still held. -pub fn waitLocked(wq: *WaitQueue) void { - const t = cur(); +pub fn waitLocked(wait_queue: *WaitQueue) void { + const t = current(); t.state = .blocked; - t.next = wq.head; - wq.head = t; + t.next = wait_queue.head; + wait_queue.head = t; schedule(); } -/// Move the highest-priority waiter on `wq` (if any) to the ready queue. +/// Move the highest-priority waiter on `wait_queue` (if any) to the ready queue. /// Precondition: the big kernel lock is held. Does not preempt — the caller decides. -pub fn wakeLocked(wq: *WaitQueue) void { +pub fn wakeLocked(wait_queue: *WaitQueue) void { // Find the highest-priority waiter (bounded scan) and unlink it. - var best_prev: ?*Task = null; + var best_previous: ?*Task = null; var best: ?*Task = null; - var prev: ?*Task = null; - var node = wq.head; + var previous: ?*Task = null; + var node = wait_queue.head; while (node) |t| : ({ - prev = t; + previous = t; node = t.next; }) { if (best == null or t.priority > best.?.priority) { best = t; - best_prev = prev; + best_previous = previous; } } const t = best orelse return; - if (best_prev) |p| p.next = t.next else wq.head = t.next; + if (best_previous) |p| p.next = t.next else wait_queue.head = t.next; t.state = .ready; enqueue(t); } @@ -424,7 +424,7 @@ pub fn wakeLocked(wq: *WaitQueue) void { /// FIFO). Precondition: the big kernel lock is held; still held on return (when the /// task is made ready again). The IPC layer's counterpart to `waitLocked`. pub fn blockCurrentLocked() void { - cur().state = .blocked; + current().state = .blocked; schedule(); } @@ -436,18 +436,18 @@ pub fn readyLocked(t: *Task) void { enqueue(t); } -/// Block on `wq` (a self-contained critical section). -pub fn wait(wq: *WaitQueue) void { +/// Block on `wait_queue` (a self-contained critical section). +pub fn wait(wait_queue: *WaitQueue) void { const flags = sync.enter(); - waitLocked(wq); + waitLocked(wait_queue); sync.leave(flags); } -/// Wake the highest-priority waiter on `wq`, preempting if it outranks us. -pub fn wake(wq: *WaitQueue) void { +/// Wake the highest-priority waiter on `wait_queue`, preempting if it outranks us. +pub fn wake(wait_queue: *WaitQueue) void { const flags = sync.enter(); const pc = thisCpu(); - wakeLocked(wq); + wakeLocked(wait_queue); // If a task this core would now pick outranks the running one, run it at once. // (A waiter pinned to *another* core isn't counted — that core picks it up on its // next tick; this core doesn't preempt for work it can't run.) @@ -468,7 +468,7 @@ fn highestReadyPriority(pc: *PerCpu) ?Priority { /// Wake any sleeping task whose deadline has passed. Bounded by the task count, /// so it stays deterministic. Called from the timer tick (interrupts disabled). fn wakeExpired() void { - const now = arch.millis(); + const now = architecture.millis(); for (&tasks) |*t| { if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) { t.wake_at = 0; @@ -521,10 +521,10 @@ pub fn exitUser() noreturn { const dying = pc.current; const as = dying.aspace; if (as != 0) { - const kroot = arch.kernelPageTable(); - arch.loadPageTable(kroot); // off the process tables before freeing them + const kroot = architecture.kernelPageTable(); + architecture.loadPageTable(kroot); // off the process tables before freeing them pc.loaded_aspace = kroot; - arch.destroyAddressSpace(as); + architecture.destroyAddressSpace(as); } dying.state = .free; dying.aspace = 0; @@ -538,11 +538,11 @@ pub fn exitUser() noreturn { /// Whether the running task is a user process (has its own address space). pub fn currentIsUserProcess() bool { - return cur().aspace != 0; + return current().aspace != 0; } pub fn currentId() u32 { - return cur().id; + return current().id; } /// The dense index of the core this task is currently running on (0 = BSP). Reads @@ -551,11 +551,11 @@ pub fn currentId() u32 { /// base isn't published yet (a fault in very early boot, before `init`), so a fault /// reporter can call it unconditionally without a second fault. pub fn currentCpuIndex() u32 { - if (arch.cpuLocal() == 0) return 0; + if (architecture.cpuLocal() == 0) return 0; return thisCpu().index; } /// Change the running task's priority (takes effect next time it's enqueued). pub fn setPriority(p: Priority) void { - cur().priority = p; + current().priority = p; } diff --git a/src/kernel/sync.zig b/src/kernel/sync.zig index ffa03f8..ab2699d 100644 --- a/src/kernel/sync.zig +++ b/src/kernel/sync.zig @@ -28,7 +28,7 @@ //! `releaseForFreshTask` before running the task body. const std = @import("std"); -const arch = @import("arch"); +const architecture = @import("architecture"); /// 0 = free, 1 = held. A single global lock for the whole kernel. var held = std.atomic.Value(u32).init(0); @@ -38,7 +38,7 @@ var held = std.atomic.Value(u32).init(0); /// Interrupts stay off for the whole critical section so this core's timer tick /// can't try to re-acquire the lock we're holding. pub fn enter() u64 { - const flags = arch.saveInterrupts(); + const flags = architecture.saveInterrupts(); acquire(); return flags; } @@ -48,7 +48,7 @@ pub fn enter() u64 { /// section reached from task context (`yield`, `sleep`, `wait`, `wake`, IPC). pub fn leave(flags: u64) void { release(); - arch.restoreInterrupts(flags); + architecture.restoreInterrupts(flags); } /// Release the lock but leave interrupts as they are. The exit for a critical @@ -71,7 +71,7 @@ fn acquire() void { // Test-and-test-and-set: try once, then spin read-only until the lock looks // free before retrying the (bus-locked) swap — cheaper on the coherency fabric. while (held.swap(1, .acquire) != 0) { - while (held.load(.monotonic) != 0) arch.cpuRelax(); + while (held.load(.monotonic) != 0) architecture.cpuRelax(); } } diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index 3bbcf66..1df4f0b 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -11,20 +11,23 @@ const std = @import("std"); const danos = @import("danos"); -const arch = @import("arch"); +const architecture = @import("architecture"); +const device_service = @import("device-service.zig"); const platform = @import("platform"); const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); -const sched = @import("scheduler.zig"); +const scheduler = @import("scheduler.zig"); const ipc = @import("ipc.zig"); -const ipcsync = @import("ipc_sync.zig"); +const ipcsync = @import("ipc-synchronous.zig"); +const irq = @import("irq.zig"); +const sync = @import("sync.zig"); const process = @import("process.zig"); const initrd = @import("initrd"); /// Formatted write straight to serial, independent of the framebuffer console. fn log(comptime fmt: []const u8, args: anytype) void { - var buf: [128]u8 = undefined; - arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return); + var buffer: [128]u8 = undefined; + architecture.serialWrite(std.fmt.bufPrint(&buffer, fmt, args) catch return); } var passed: u32 = 0; @@ -50,9 +53,9 @@ fn result() void { log("DANOS-TEST-DONE\n", .{}); } -pub fn run(case: []const u8, boot_info: *const BootInfo) void { +pub fn run(case: []const u8, boot_information: *const BootInformation) void { if (eql(case, "smoke")) { - smoke(boot_info); + smoke(boot_information); } else if (eql(case, "discovery")) { discoveryTest(); } else if (eql(case, "wx")) { @@ -66,7 +69,7 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { } else if (eql(case, "heap")) { heapTest(); } else if (eql(case, "sched")) { - schedTest(); + schedulerTest(); } else if (eql(case, "priority")) { priorityTest(); } else if (eql(case, "sleep")) { @@ -102,17 +105,21 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { } else if (eql(case, "user-pf")) { userPfTest(); } else if (eql(case, "init")) { - initTest(boot_info); + initTest(boot_information); } else if (eql(case, "process")) { - processTest(boot_info); + processTest(boot_information); } else if (eql(case, "initrd")) { - initrdTest(boot_info); + initrdTest(boot_information); } else if (eql(case, "vfs")) { - vfsTest(boot_info); + vfsTest(boot_information); } else if (eql(case, "hpet")) { - hpetTest(boot_info); + hpetTest(boot_information); } else if (eql(case, "iopass")) { ioPassTest(); + } else if (eql(case, "irqfree")) { + irqFreeTest(); + } else if (eql(case, "bus")) { + busTest(boot_information); } else if (eql(case, "poweroff")) { powerTest(.off); } else if (eql(case, "reboot")) { @@ -124,9 +131,9 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { fn platformHal() platform.Hal { return .{ - .mapMmio = arch.mapMmio, - .pioRead = arch.pioRead, - .pioWrite = arch.pioWrite, + .mapMmio = architecture.mapMmio, + .pioRead = architecture.pioRead, + .pioWrite = architecture.pioWrite, }; } @@ -146,19 +153,19 @@ fn powerTest(comptime action: enum { off, reboot }) void { result(); } -const BootInfo = danos.BootInfo; +const BootInformation = danos.BootInformation; fn eql(a: []const u8, b: []const u8) bool { return std.mem.eql(u8, a, b); } /// Non-destructive checks of the memory map and frame allocator. -fn smoke(boot_info: *const BootInfo) void { +fn smoke(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: smoke\n", .{}); // The memory map has some usable RAM. - const mm = boot_info.memory_map; - const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len]; + const mm = boot_information.memory_map; + const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len]; var usable: u64 = 0; for (regions) |r| { if (r.kind == .usable) usable += r.pages; @@ -179,7 +186,7 @@ fn smoke(boot_info: *const BootInfo) void { check("free returns frames to the pool", pmm.stats().free_frames == before + 2); // Paging is active on our own tables (the root is non-zero and page-aligned). - const root = arch.activePageTable(); + const root = architecture.activePageTable(); check("paging active (page-table root set)", root != 0 and root % danos.page_size == 0); result(); @@ -189,13 +196,13 @@ fn smoke(boot_info: *const BootInfo) void { /// on its own. Interrupts are already enabled by kmain before tests run. fn timer() void { log("DANOS-TEST-BEGIN: timer\n", .{}); - const start = arch.ticks(); - // Busy-wait for the counter to advance. arch.ticks() is a volatile load, so + const start = architecture.ticks(); + // Busy-wait for the counter to advance. architecture.ticks() is a volatile load, so // the compiler re-reads it each iteration and sees the interrupt's update. // The cap is only a safety net; the harness timeout is the real backstop. var spins: u64 = 0; - while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1; - check("timer interrupts advance the tick count", arch.ticks() > start); + while (architecture.ticks() == start and spins < 5_000_000_000) spins +%= 1; + check("timer interrupts advance the tick count", architecture.ticks() > start); result(); } @@ -207,8 +214,8 @@ fn timer() void { /// (the sleep type, plus the integrity check that every byte was consumed). fn discoveryTest() void { log("DANOS-TEST-BEGIN: discovery\n", .{}); - const pinfo = platform.platformInfo(); - const pw = platform.powerInfo(); + const pinfo = platform.platformInformation(); + const pw = platform.powerInformation(); const am = platform.amlStats(); check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000); @@ -223,27 +230,27 @@ fn discoveryTest() void { } /// Audit the W^X invariant across the memory classes: kernel code must be -/// executable, everything else must not be. `arch.pageExecutable` reads the leaf +/// executable, everything else must not be. `architecture.pageExecutable` reads the leaf /// page-table entry's NX bit, so this guards the permission overlay in paging.zig — /// a broader check than `fault-nx`, which only exercises one data page. fn wxTest() void { log("DANOS-TEST-BEGIN: wx\n", .{}); - check("kernel code is executable (R+X)", arch.pageExecutable(@intFromPtr(&wxTest))); + check("kernel code is executable (R+X)", architecture.pageExecutable(@intFromPtr(&wxTest))); const ro = "danos-wx-probe"; // string literal -> .rodata - check("rodata is non-executable (NX)", !arch.pageExecutable(@intFromPtr(ro.ptr))); + check("rodata is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(ro.ptr))); - check("kernel data is non-executable (NX)", !arch.pageExecutable(@intFromPtr(&passed))); + check("kernel data is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(&passed))); if (heap.allocator().alloc(u8, 64) catch null) |h| { - check("heap is non-executable (NX)", !arch.pageExecutable(@intFromPtr(h.ptr))); + check("heap is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(h.ptr))); heap.allocator().free(h); } var local: u64 = 0; _ = &local; - check("stack is non-executable (NX)", !arch.pageExecutable(@intFromPtr(&local))); + check("stack is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(&local))); result(); } @@ -254,15 +261,15 @@ fn vmm() void { log("DANOS-TEST-BEGIN: vmm\n", .{}); const frame = pmm.alloc(); check("frame available to map", frame != null); - if (frame) |phys| { - var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped - arch.mapPage(virt, phys, true); - const p: *volatile u64 = @ptrFromInt(virt); + if (frame) |physical| { + var virtual: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped + architecture.mapPage(virtual, physical, true); + const p: *volatile u64 = @ptrFromInt(virtual); p.* = 0xdead_c0de_cafe_babe; check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe); - arch.unmapPage(virt); - pmm.free(phys); - virt += 0; + architecture.unmapPage(virtual); + pmm.free(physical); + virtual += 0; } result(); } @@ -274,9 +281,9 @@ fn heapTest() void { const a = heap.allocator(); // Allocate, write a pattern, read it back, free. - const buf = a.alloc(u8, 4096) catch null; - check("alloc 4096 bytes", buf != null); - if (buf) |b| { + const buffer = a.alloc(u8, 4096) catch null; + check("alloc 4096 bytes", buffer != null); + if (buffer) |b| { @memset(b, 0xAB); check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB); a.free(b); @@ -284,11 +291,11 @@ fn heapTest() void { // Freeing then re-allocating the same size should reuse the block. const p1 = a.alloc(u64, 8) catch null; - const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0; + const address1 = if (p1) |p| @intFromPtr(p.ptr) else 0; if (p1) |p| a.free(p); const p2 = a.alloc(u64, 8) catch null; - const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0; - check("freed block is reused", addr1 != 0 and addr1 == addr2); + const address2 = if (p2) |p| @intFromPtr(p.ptr) else 0; + check("freed block is reused", address1 != 0 and address1 == address2); if (p2) |p| a.free(p); // Force growth past the initial page and check every block is usable. @@ -336,32 +343,32 @@ fn heapTest() void { fn clock() void { log("DANOS-TEST-BEGIN: clock\n", .{}); - const timer_clock = arch.timerClockHz(); + const timer_clock = architecture.timerClockHz(); check("timer clock frequency measured", timer_clock > 1_000_000 and timer_clock < 100_000_000_000); - const clock_hz = arch.clockHz(); + const clock_hz = architecture.clockHz(); check("monotonic clock frequency measured", clock_hz > 100_000_000 and clock_hz < 100_000_000_000); // Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms). - const start_ticks = arch.ticks(); - const start_ms = arch.millis(); + const start_ticks = architecture.ticks(); + const start_ms = architecture.millis(); var spins: u64 = 0; - while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1; - const elapsed_ms = arch.millis() - start_ms; + while (architecture.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1; + const elapsed_ms = architecture.millis() - start_ms; check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100); // Sub-millisecond resolution: spin until nanos() first advances, then confirm // that first step happened within a millisecond — so nanos() resolves finer // than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the // first change is robust to QEMU's coarse TSC update granularity. - const n1 = arch.nanos(); + const n1 = architecture.nanos(); var s2: u64 = 0; - while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1; - const n2 = arch.nanos(); + while (architecture.nanos() == n1 and s2 < 10_000_000) s2 +%= 1; + const n2 = architecture.nanos(); check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000); // The unit functions agree (within rounding). - const ns = arch.nanos(); - check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2)); + const ns = architecture.nanos(); + check("nanos/micros/millis are consistent", diffWithin(architecture.micros(), ns / 1000, 1000) and diffWithin(architecture.millis(), ns / 1_000_000, 2)); result(); } @@ -390,12 +397,12 @@ fn spin2() void { /// Preemption: spawn three tasks that busy-loop *without* yielding. If they all /// make progress, the timer must be preempting between them (and the context /// switch works) — because nothing yields voluntarily. -fn schedTest() void { +fn schedulerTest() void { log("DANOS-TEST-BEGIN: sched\n", .{}); counters = .{ 0, 0, 0 }; - sched.spawn(spin0, 4); - sched.spawn(spin1, 4); - sched.spawn(spin2, 4); + scheduler.spawn(spin0, 4); + scheduler.spawn(spin1, 4); + scheduler.spawn(spin2, 4); const c0: *volatile u64 = &counters[0]; const c1: *volatile u64 = &counters[1]; @@ -413,7 +420,7 @@ var run_n: usize = 0; fn recordExit(priority: u8) void { run_order[run_n] = priority; run_n += 1; - sched.exit(); + scheduler.exit(); } fn taskHigh() void { recordExit(6); @@ -429,31 +436,31 @@ fn taskLow() void { /// priorities and let them run cooperatively. They must run highest-first. fn priorityTest() void { log("DANOS-TEST-BEGIN: priority\n", .{}); - sched.setPreemption(false); - sched.setPriority(1); // above the idle task (0), below the workers — runs last + scheduler.setPreemption(false); + scheduler.setPriority(1); // above the idle task (0), below the workers — runs last run_n = 0; - sched.spawn(taskLow, 2); - sched.spawn(taskMid, 4); - sched.spawn(taskHigh, 6); + scheduler.spawn(taskLow, 2); + scheduler.spawn(taskMid, 4); + scheduler.spawn(taskHigh, 6); - while (run_n < 3) sched.yield(); // regain control only once the workers are done + while (run_n < 3) scheduler.yield(); // regain control only once the workers are done check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2); - sched.setPriority(4); - sched.setPreemption(true); + scheduler.setPriority(4); + scheduler.setPreemption(true); result(); } -var event_wq: sched.WaitQueue = .{}; +var event_wait_queue: scheduler.WaitQueue = .{}; var event_stage: u32 = 0; fn eventWaiter() void { event_stage = 1; // reached the wait - sched.wait(&event_wq); // block until woken + scheduler.wait(&event_wait_queue); // block until woken event_stage = 3; // woken and resumed - sched.exit(); + scheduler.exit(); } /// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is @@ -461,51 +468,51 @@ fn eventWaiter() void { fn eventTest() void { log("DANOS-TEST-BEGIN: event\n", .{}); event_stage = 0; - sched.spawn(eventWaiter, 6); // higher priority than this task (4) + scheduler.spawn(eventWaiter, 6); // higher priority than this task (4) var spins: u64 = 0; - while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield(); + while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) scheduler.yield(); check("waiter reached the wait and blocked", event_stage == 1); - sched.wake(&event_wq); + scheduler.wake(&event_wait_queue); check("wake resumed the blocked waiter (preempting)", event_stage == 3); result(); } var channel: ipc.Channel(u64, 4) = .{}; -var recv_sum: u64 = 0; -var recv_count: u64 = 0; +var receive_sum: u64 = 0; +var receive_count: u64 = 0; fn producer() void { var i: u64 = 1; while (i <= 100) : (i += 1) channel.send(i); - sched.exit(); + scheduler.exit(); } fn consumer() void { var n: u64 = 0; while (n < 100) : (n += 1) { - recv_sum += channel.recv(); - recv_count += 1; + receive_sum += channel.receive(); + receive_count += 1; } - sched.exit(); + scheduler.exit(); } /// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The /// small buffer forces the channel full and empty repeatedly, exercising both the -/// blocking-send and blocking-recv paths. The messages must arrive intact. +/// blocking-send and blocking-receive paths. The messages must arrive intact. fn ipcTest() void { log("DANOS-TEST-BEGIN: ipc\n", .{}); channel = .{}; - recv_sum = 0; - recv_count = 0; - sched.spawn(consumer, 5); // above this task (4) so they run and we observe after - sched.spawn(producer, 5); + receive_sum = 0; + receive_count = 0; + scheduler.spawn(consumer, 5); // above this task (4) so they run and we observe after + scheduler.spawn(producer, 5); var spins: u64 = 0; - while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield(); + while (receive_count < 100 and spins < 2_000_000_000) : (spins += 1) scheduler.yield(); - check("all 100 messages received", recv_count == 100); - check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050); + check("all 100 messages received", receive_count == 100); + check("messages arrived intact (sum 1..100 == 5050)", receive_sum == 5050); result(); } @@ -513,9 +520,9 @@ fn ipcTest() void { /// calibrated clock) — not busy-wait — while the idle task runs. fn sleepTest() void { log("DANOS-TEST-BEGIN: sleep\n", .{}); - const t0 = arch.millis(); - sched.sleep(50); - const elapsed = arch.millis() - t0; + const t0 = architecture.millis(); + scheduler.sleep(50); + const elapsed = architecture.millis() - t0; check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70); result(); } @@ -530,10 +537,10 @@ var smp_running: bool = true; fn smpWorker() void { const p: *volatile bool = &smp_running; while (p.*) { - const c = sched.currentCpuIndex(); + const c = scheduler.currentCpuIndex(); if (c < seen_core.len) seen_core[c] = true; } - sched.exit(); + scheduler.exit(); } /// Prove tasks run **in parallel** on multiple cores (not just interleaved on one). @@ -547,7 +554,7 @@ fn smpTest() void { smp_running = true; var i: usize = 0; - while (i < 4) : (i += 1) sched.spawn(smpWorker, 4); + while (i < 4) : (i += 1) scheduler.spawn(smpWorker, 4); // Let the workers run across cores for a stretch of real time. var spins: u64 = 0; @@ -563,16 +570,16 @@ fn smpTest() void { // Bring-up is done, so the trampoline frame must be inert: zeroed (no stale code) // and non-executable (W^X restored). It's armed only while a core is climbing. - const tramp = arch.trampolinePage(); + const tramp = architecture.trampolinePage(); check("trampoline frame reserved", tramp != 0); if (tramp != 0) { - const bytes: [*]const u8 = @ptrFromInt(danos.physToVirt(tramp)); + const bytes: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(tramp)); var zeroed = true; for (0..4096) |b| { if (bytes[b] != 0) zeroed = false; } check("trampoline page zeroed when dormant", zeroed); - check("trampoline page non-executable when dormant", !arch.pageExecutable(tramp)); + check("trampoline page non-executable when dormant", !architecture.pageExecutable(tramp)); } result(); } @@ -585,10 +592,10 @@ var affinity_running: bool = true; fn affinityWorker() void { const p: *volatile bool = &affinity_running; while (p.*) { - const c = sched.currentCpuIndex(); + const c = scheduler.currentCpuIndex(); if (c < affinity_cores.len) affinity_cores[c] = true; } - sched.exit(); + scheduler.exit(); } /// A task pinned to a core must run **only** on that core. Pin a busy worker to @@ -601,7 +608,7 @@ fn affinityTest() void { affinity_cores = .{false} ** 8; affinity_running = true; - if (!sched.spawnOn(affinityWorker, 4, 1)) { + if (!scheduler.spawnOn(affinityWorker, 4, 1)) { check("worker pinned to core 1 (run with -smp)", false); result(); return; @@ -630,7 +637,7 @@ const stress_msgs = 100_000; // messages per pair const stress_cap = 4; // small channel -> constant block/wake, more lock churn var stress_chan = [_]ipc.Channel(u64, stress_cap){.{}} ** stress_pairs; -var stress_recv = [_]u64{0} ** stress_pairs; // messages received per pair +var stress_receive = [_]u64{0} ** stress_pairs; // messages received per pair var stress_order_ok = [_]bool{true} ** stress_pairs; // FIFO order held per pair var stress_cores = [_]bool{false} ** 8; // cores that ran a consumer var stress_prod_claim: usize = 0; @@ -638,27 +645,27 @@ var stress_cons_claim: usize = 0; fn stressProducer() void { // Claim a unique pair index (atomic: producers start on different cores). - const idx = @atomicRmw(usize, &stress_prod_claim, .Add, 1, .monotonic); + const index = @atomicRmw(usize, &stress_prod_claim, .Add, 1, .monotonic); var v: u64 = 1; - while (v <= stress_msgs) : (v += 1) stress_chan[idx].send(v); - sched.exit(); + while (v <= stress_msgs) : (v += 1) stress_chan[index].send(v); + scheduler.exit(); } fn stressConsumer() void { - const idx = @atomicRmw(usize, &stress_cons_claim, .Add, 1, .monotonic); + const index = @atomicRmw(usize, &stress_cons_claim, .Add, 1, .monotonic); var expected: u64 = 1; while (expected <= stress_msgs) : (expected += 1) { - const got = stress_chan[idx].recv(); - if (got != expected) stress_order_ok[idx] = false; // lost/reordered => lock broke - const c = sched.currentCpuIndex(); + const got = stress_chan[index].receive(); + if (got != expected) stress_order_ok[index] = false; // lost/reordered => lock broke + const c = scheduler.currentCpuIndex(); if (c < stress_cores.len) stress_cores[c] = true; - stress_recv[idx] = expected; + stress_receive[index] = expected; } - sched.exit(); + scheduler.exit(); } /// Stress the big kernel lock under sustained cross-core contention. Each pair drives -/// `stress_msgs` sequenced messages through a 4-slot channel — every send and recv +/// `stress_msgs` sequenced messages through a 4-slot channel — every send and receive /// takes the lock, and the small buffer forces constant block/wake (so the scheduler /// churns too). A single-producer/single-consumer channel must deliver in strict FIFO /// order; if the lock let two cores into a critical section at once, the ring buffer @@ -667,32 +674,32 @@ fn stressConsumer() void { fn stressTest() void { log("DANOS-TEST-BEGIN: smp-stress\n", .{}); stress_chan = [_]ipc.Channel(u64, stress_cap){.{}} ** stress_pairs; - stress_recv = [_]u64{0} ** stress_pairs; + stress_receive = [_]u64{0} ** stress_pairs; stress_order_ok = [_]bool{true} ** stress_pairs; stress_cores = [_]bool{false} ** 8; stress_prod_claim = 0; stress_cons_claim = 0; var i: usize = 0; - while (i < stress_pairs) : (i += 1) sched.spawn(stressConsumer, 4); + while (i < stress_pairs) : (i += 1) scheduler.spawn(stressConsumer, 4); i = 0; - while (i < stress_pairs) : (i += 1) sched.spawn(stressProducer, 4); + while (i < stress_pairs) : (i += 1) scheduler.spawn(stressProducer, 4); // Drop below the workers so they get the cores; wake periodically to check for // completion. A broken lock instead hangs here (harness timeout) or faults. - sched.setPriority(1); + scheduler.setPriority(1); var spins: u64 = 0; while (spins < 40_000_000_000) : (spins += 1) { var done = true; - for (stress_recv) |n| { + for (stress_receive) |n| { if (n < stress_msgs) done = false; } if (done) break; } - sched.setPriority(4); + scheduler.setPriority(4); var total: u64 = 0; - for (stress_recv) |n| total += n; + for (stress_receive) |n| total += n; var order_ok = true; for (stress_order_ok) |ok| { if (!ok) order_ok = false; @@ -709,14 +716,14 @@ fn stressTest() void { result(); } -/// Retry: `main` forced the first AP wake attempt to fail (arch.testFailNextWakes), +/// Retry: `main` forced the first AP wake attempt to fail (architecture.testFailNextWakes), /// so a core missed its first INIT-SIPI-SIPI. The boot retry must have brought it back /// anyway — every enumerated core should be online. If retry were broken, that core /// would be parked and the count would fall short. fn smpRetryTest() void { log("DANOS-TEST-BEGIN: smp-retry\n", .{}); const total = platform.cpus().len; - const online = sched.onlineCount(); + const online = scheduler.onlineCount(); log("DANOS-RETRY: {d}/{d} cores online after a forced first-wake failure\n", .{ online, total }); check("multiple cores enumerated (run with -smp)", total >= 2); check("retry brought every core online despite a failed first wake", online == total); @@ -726,7 +733,7 @@ fn smpRetryTest() void { // --- ring 3 (user mode) ----------------------------------------------------- /// The mmap/munmap grant path: create a fresh address space, hand out pages into -/// its mmap arena the way the `mmap` syscall does, prove they are real (write and +/// its mmap arena the way the `mmap` system_call does, prove they are real (write and /// read them back through the physmap), then release them via the `munmap` path /// (translate -> unmap -> free) and tear the address space down. The frame count /// must return exactly to where it started — a leak or a double-free would show @@ -736,7 +743,7 @@ fn userMemTest() void { log("DANOS-TEST-BEGIN: usermem\n", .{}); const base_free = pmm.stats().free_frames; - const aspace = arch.createAddressSpace() orelse { + const aspace = architecture.createAddressSpace() orelse { check("created a fresh address space", false); result(); return; @@ -750,7 +757,7 @@ fn userMemTest() void { var mapped: usize = 0; while (mapped < npages) : (mapped += 1) { frames[mapped] = pmm.alloc() orelse break; - arch.mapUserPageInto(aspace, arena + mapped * danos.page_size, frames[mapped], true, false); + architecture.mapUserPageInto(aspace, arena + mapped * danos.page_size, frames[mapped], true, false); } check("granted three user pages", mapped == npages); @@ -759,12 +766,12 @@ fn userMemTest() void { var rw_ok = true; for (0..npages) |i| { const va = arena + i * danos.page_size; - const phys = arch.translate(aspace, va) orelse { + const physical = architecture.translate(aspace, va) orelse { translate_ok = false; continue; }; - if (phys != frames[i]) translate_ok = false; - const p: [*]u8 = @ptrFromInt(danos.physToVirt(phys)); + if (physical != frames[i]) translate_ok = false; + const p: [*]u8 = @ptrFromInt(danos.physicalToVirtual(physical)); p[0] = 0xA5; if (p[0] != 0xA5) rw_ok = false; } @@ -774,13 +781,13 @@ fn userMemTest() void { // Release them the way munmap does, then tear down the address space. for (0..npages) |i| { const va = arena + i * danos.page_size; - if (arch.translate(aspace, va)) |phys| { - arch.unmapUserPageInto(aspace, va); - pmm.free(phys); + if (architecture.translate(aspace, va)) |physical| { + architecture.unmapUserPageInto(aspace, va); + pmm.free(physical); } } - check("munmap unmapped every grant", arch.translate(aspace, arena) == null); - arch.destroyAddressSpace(aspace); + check("munmap unmapped every grant", architecture.translate(aspace, arena) == null); + architecture.destroyAddressSpace(aspace); check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free); result(); @@ -788,21 +795,21 @@ fn userMemTest() void { // --- synchronous IPC -------------------------------------------------------- -var ipc_ep: *ipcsync.Endpoint = undefined; +var ipc_endpoint: *ipcsync.Endpoint = undefined; var ipc_replies_ok: bool = false; var ipc_done: bool = false; /// Echo-increment server: reply to each request with request+1, forever. fn ipcServer() void { - var reply_buf: [8]u8 = undefined; + var reply_buffer: [8]u8 = undefined; var reply_len: u64 = 0; var badge: u64 = 0; while (true) { - var recv: [8]u8 = undefined; - const n = ipcsync.replyWait(ipc_ep, @intFromPtr(&reply_buf), reply_len, @intFromPtr(&recv), recv.len, &badge); - if (n < 0) sched.exit(); - const v = std.mem.readInt(u64, recv[0..8], .little); - std.mem.writeInt(u64, reply_buf[0..8], v + 1, .little); + var receive: [8]u8 = undefined; + const n = ipcsync.replyWait(ipc_endpoint, @intFromPtr(&reply_buffer), reply_len, @intFromPtr(&receive), receive.len, &badge); + if (n < 0) scheduler.exit(); + const v = std.mem.readInt(u64, receive[0..8], .little); + std.mem.writeInt(u64, reply_buffer[0..8], v + 1, .little); reply_len = 8; } } @@ -812,15 +819,15 @@ fn ipcClient() void { var ok = true; var i: u64 = 0; while (i < 100) : (i += 1) { - var msg: [8]u8 = undefined; - std.mem.writeInt(u64, msg[0..8], i, .little); + var message: [8]u8 = undefined; + std.mem.writeInt(u64, message[0..8], i, .little); var reply: [8]u8 = undefined; - const n = ipcsync.call(ipc_ep, @intFromPtr(&msg), 8, @intFromPtr(&reply), reply.len); + const n = ipcsync.call(ipc_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len); if (n != 8 or std.mem.readInt(u64, reply[0..8], .little) != i + 1) ok = false; } ipc_replies_ok = ok; ipc_done = true; - sched.exit(); + scheduler.exit(); } /// Synchronous IPC: a client and a server (two kernel tasks) ping-pong 100 calls @@ -830,15 +837,15 @@ fn ipcClient() void { /// prove the block/wake and reply-routing paths. (Kernel tasks, so no user ELF.) fn ipcCallTest() void { log("DANOS-TEST-BEGIN: ipc-call\n", .{}); - ipc_ep = ipcsync.createEndpoint().?; + ipc_endpoint = ipcsync.createEndpoint().?; ipc_replies_ok = false; ipc_done = false; - sched.spawn(ipcServer, 5); // above this task, so the workers run - sched.spawn(ipcClient, 5); + scheduler.spawn(ipcServer, 5); // above this task, so the workers run + scheduler.spawn(ipcClient, 5); const done: *volatile bool = &ipc_done; var spins: u64 = 0; - while (!done.* and spins < 100_000_000) : (spins += 1) sched.yield(); + while (!done.* and spins < 100_000_000) : (spins += 1) scheduler.yield(); check("client completed 100 synchronous calls", ipc_done); check("every reply was request+1 (rendezvous + reply routing intact)", ipc_replies_ok); @@ -854,7 +861,7 @@ fn procWorker() void { const running: *volatile bool = &proc_worker_run; const ran: *volatile bool = &proc_worker_ran; while (running.*) ran.* = true; - sched.exit(); + scheduler.exit(); } /// Real processes: load /sbin/init as TWO scheduled ring-3 processes, each with @@ -863,20 +870,20 @@ fn procWorker() void { /// only happen if each runs on its own page tables (CR3 switched correctly per /// process) and preemption interleaves them with the kernel worker. This is the /// strongest cheap proof of address-space isolation. -fn processTest(boot_info: *const BootInfo) void { +fn processTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: process\n", .{}); - check("bootloader handed over sbin/init", boot_info.init_len != 0); - if (boot_info.init_len == 0) { + check("bootloader handed over sbin/init", boot_information.init_len != 0); + if (boot_information.init_len == 0) { result(); return; } - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len]; process.write_count = 0; process.write_from_user = false; proc_worker_run = true; proc_worker_ran = false; - sched.spawn(procWorker, 4); // kernel task at the processes' priority + scheduler.spawn(procWorker, 4); // kernel task at the processes' priority var spawned: u32 = 0; if (process.spawnProcess(image, 4)) spawned += 1 else |_| {} @@ -884,10 +891,10 @@ fn processTest(boot_info: *const BootInfo) void { // Wait (real time) for several heartbeats across the two processes. Each // process sleeps ~1 s between beats, so a few seconds yields several. - sched.setPriority(1); // drop below the workers so they get the cores - const deadline = arch.millis() + 8000; - while (process.write_count < 4 and arch.millis() < deadline) sched.yield(); - sched.setPriority(4); + scheduler.setPriority(1); // drop below the workers so they get the cores + const deadline = architecture.millis() + 8000; + while (process.write_count < 4 and architecture.millis() < deadline) scheduler.yield(); + scheduler.setPriority(4); proc_worker_run = false; log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, process.write_count }); @@ -904,7 +911,7 @@ fn processTest(boot_info: *const BootInfo) void { /// read is somehow allowed the blob spins and the harness times out. fn userPfTest() void { log("DANOS-TEST-BEGIN: user-pf\n", .{}); - sched.setPreemption(false); + scheduler.setPreemption(false); _ = process.run(process.pfBlob()) catch {}; log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{}); } @@ -914,14 +921,14 @@ fn userPfTest() void { /// — the same call the normal boot path makes — then confirm it beats. init /// heartbeats forever, so this proves it reaches ring 3, makes repeated syscalls /// (write + sleep), and stays alive rather than exiting. -fn initTest(boot_info: *const BootInfo) void { +fn initTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: init\n", .{}); - check("bootloader handed over sbin/init", boot_info.init_len != 0); - if (boot_info.init_len == 0) { + check("bootloader handed over sbin/init", boot_information.init_len != 0); + if (boot_information.init_len == 0) { result(); return; } - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len]; process.write_count = 0; const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: { log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)}); @@ -931,13 +938,13 @@ fn initTest(boot_info: *const BootInfo) void { // Wait (real time) for at least two heartbeats — proving it runs, writes, // and sleeps repeatedly (init sleeps ~1 s between beats). - sched.setPriority(1); - const deadline = arch.millis() + 8000; - while (process.write_count < 2 and arch.millis() < deadline) sched.yield(); - sched.setPriority(4); + scheduler.setPriority(1); + const deadline = architecture.millis() + 8000; + while (process.write_count < 2 and architecture.millis() < deadline) scheduler.yield(); + scheduler.setPriority(4); const prefix = "init: heartbeat"; - const beat_ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix); + const beat_ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix); check("init produced repeated heartbeats (>=2)", process.write_count >= 2); check("heartbeat text arrived intact", beat_ok); check("heartbeats came from user mode (CPL 3)", process.write_from_user); @@ -948,14 +955,14 @@ fn initTest(boot_info: *const BootInfo) void { /// The initrd path: the bootloader handed over an image bundling extra user /// binaries; parse it, spawn every program, and confirm one (the vfs stub) /// reaches ring 3 and heartbeats — proving the whole ferry-parse-spawn pipeline. -fn initrdTest(boot_info: *const BootInfo) void { +fn initrdTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: initrd\n", .{}); - check("bootloader handed over an initrd", boot_info.initrd_len != 0); - if (boot_info.initrd_len == 0) { + check("bootloader handed over an initrd", boot_information.initrd_len != 0); + if (boot_information.initrd_len == 0) { result(); return; } - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.initrd_base)))[0..boot_info.initrd_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len]; const rd = initrd.Reader.init(image) orelse { check("initrd image is valid", false); result(); @@ -977,10 +984,10 @@ fn initrdTest(boot_info: *const BootInfo) void { check("every initrd binary spawned", spawned == rd.count); // Wait for the spawned programs to run and make syscalls (they write + sleep). - sched.setPriority(1); - const deadline = arch.millis() + 8000; - while (process.write_count < 2 and arch.millis() < deadline) sched.yield(); - sched.setPriority(4); + scheduler.setPriority(1); + const deadline = architecture.millis() + 8000; + while (process.write_count < 2 and architecture.millis() < deadline) scheduler.yield(); + scheduler.setPriority(4); check("initrd processes ran and made syscalls (>=2)", process.write_count >= 2); check("syscalls came from user mode (CPL 3)", process.write_from_user); @@ -988,18 +995,18 @@ fn initrdTest(boot_info: *const BootInfo) void { } /// The full VFS path: spawn the user-space VFS server and a client from the -/// initrd. The client opens a file through the rt file API, writes, seeks, reads +/// initrd. The client opens a file through the runtime file API, writes, seeks, reads /// it back, and — only if the round trip matched — heartbeats "vfstest: ok". So /// seeing that marker proves client open/write/read reached the server over IPC /// and came back correct. (The client retries until the server registers.) -fn vfsTest(boot_info: *const BootInfo) void { +fn vfsTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: vfs\n", .{}); - if (boot_info.initrd_len == 0) { + if (boot_information.initrd_len == 0) { check("bootloader handed over an initrd", false); result(); return; } - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.initrd_base)))[0..boot_info.initrd_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len]; const rd = initrd.Reader.init(image) orelse { check("initrd image is valid", false); result(); @@ -1011,19 +1018,19 @@ fn vfsTest(boot_info: *const BootInfo) void { // Spawn just the server and its client (other initrd binaries would write to // the shared evidence buffer and confuse the marker check). _ = spawnNamed(rd, "vfs"); - _ = spawnNamed(rd, "vfstest"); + _ = spawnNamed(rd, "vfs-test"); // Wait for the client's success heartbeat (it round-trips, then beats ~1/s). const prefix = "vfstest: ok"; - sched.setPriority(1); - const deadline = arch.millis() + 10000; - while (arch.millis() < deadline) { - if (process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix) and process.write_count >= 2) break; - sched.yield(); + scheduler.setPriority(1); + const deadline = architecture.millis() + 10000; + while (architecture.millis() < deadline) { + if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break; + scheduler.yield(); } - sched.setPriority(4); + scheduler.setPriority(4); - const ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix); + const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix); check("client completed the VFS round trip (open/write/read matched)", ok); check("the round trip ran repeatedly (server stays up)", process.write_count >= 2); check("client syscalls came from user mode (CPL 3)", process.write_from_user); @@ -1043,19 +1050,26 @@ fn spawnNamed(rd: initrd.Reader, name: []const u8) bool { return false; } -/// IO passthrough: a user-space driver reads real hardware. Spawn hpetd, which -/// enumerates the device table, claims the HPET, maps its MMIO registers into its -/// own ring-3 address space, enables the counter, and reads it — heartbeating -/// "hpetd: ok" only if the counter advanced. Seeing that proves a user process -/// drove real hardware through a kernel-granted MMIO mapping. -fn hpetTest(boot_info: *const BootInfo) void { +/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is +/// *woken by it*. Spawn hpetd, which claims the HPET, maps its registers into its +/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt +/// to an IPC endpoint, and then blocks. It prints "hpetd: ok" only after being woken +/// `target_ticks` times — it cannot reach that line by polling, because the loop's +/// only exit is through `replyWait` returning a notification badge. +/// +/// The interesting assertion is the last one, which doesn't trust hpetd at all: it +/// reads the I/O APIC's redirection entry back and checks the kernel really routed +/// the line (our vector, level-triggered) and really left it unmasked after the +/// driver's final `irq_ack`. hpetd disables its comparator on the last interrupt, so +/// that state is quiescent and not a race. +fn hpetTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: hpet\n", .{}); - if (boot_info.initrd_len == 0) { + if (boot_information.initrd_len == 0) { check("bootloader handed over an initrd", false); result(); return; } - const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.initrd_base)))[0..boot_info.initrd_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len]; const rd = initrd.Reader.init(image) orelse { check("initrd image is valid", false); result(); @@ -1067,20 +1081,213 @@ fn hpetTest(boot_info: *const BootInfo) void { check("hpetd spawned from the initrd", spawnNamed(rd, "hpetd")); const prefix = "hpetd: ok"; - sched.setPriority(1); - const deadline = arch.millis() + 10000; - while (arch.millis() < deadline) { - if (process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix) and process.write_count >= 2) break; - sched.yield(); + scheduler.setPriority(1); + const deadline = architecture.millis() + 10000; + while (architecture.millis() < deadline) { + if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break; + scheduler.yield(); } - sched.setPriority(4); + scheduler.setPriority(4); - const ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix); - check("user driver mapped HPET MMIO and read the counter advancing", ok); + const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix); + check("user driver mapped HPET MMIO and was woken by its interrupt", ok); check("driver syscalls came from user mode (CPL 3)", process.write_from_user); + check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk()); result(); } +/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the +/// kernel programmed it: a vector in the device window, level-triggered, unmasked. +/// Independent of anything the driver reported about itself. +fn hpetRouteOk() bool { + const gsi = hpetGsi() orelse return false; + if (gsi >= architecture.irqRouteCount()) return false; + const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0) + const vector: u8 = @truncate(low & 0xFF); + const masked = low & (1 << 16) != 0; + const level = low & (1 << 15) != 0; + return vector >= architecture.irq_vector_base and + vector < architecture.irq_vector_base + architecture.irq_vector_count and + level and !masked; +} + +/// The GSI discovery recorded for the HPET, from the same device table the driver saw. +fn hpetGsi() ?u32 { + var buffer: [16]danos.DeviceDescriptor = undefined; + const n = @min(device_service.enumerate(&buffer), buffer.len); + for (buffer[0..n]) |d| { + if (d.class != @intFromEnum(danos.DeviceClass.timer)) continue; + if (d.parent != danos.no_parent) continue; // the block, not a comparator child + for (0..d.resource_count) |j| { + const r = d.resources[j]; + if (r.kind == @intFromEnum(danos.ResourceKind.irq)) return @intCast(r.start); + } + } + return null; +} + +/// Bus driver: a user process claims a device that contains other devices, enumerates +/// them from the hardware, and publishes each as a child via `device_register` — the +/// primitive a PCI bridge or USB hub driver is built from. +/// +/// `busd` treats the HPET's register block as a bus and its comparators as children, +/// giving each a 0x20 sub-window. It checks its own work (children come back from the +/// table with the right parent and a strictly narrower window) and, importantly, that +/// the kernel **refuses** a child whose window escapes the parent's — without that, +/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints +/// "busd: ok" only if all of that holds. +/// +/// The kernel-side check here is the one busd can't make: that the children really did +/// land in the device table with the containment invariant intact. +fn busTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: bus\n", .{}); + if (boot_information.initrd_len == 0) { + check("bootloader handed over an initrd", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len]; + const rd = initrd.Reader.init(image) orelse { + check("initrd image is valid", false); + result(); + return; + }; + + process.write_count = 0; + process.write_from_user = false; + check("busd spawned from the initrd", spawnNamed(rd, "busd")); + + const prefix = "busd: ok"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 10000; + while (architecture.millis() < deadline) { + if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break; + scheduler.yield(); + } + scheduler.setPriority(4); + + const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix); + check("bus driver published children and the kernel refused an out-of-window one", ok); + check("driver syscalls came from user mode (CPL 3)", process.write_from_user); + check("every registered child is contained in its parent", childrenContained()); + result(); +} + +/// Every child `busd` registered must have each of its resources inside a parent +/// resource of the same kind — the invariant `device_register` exists to maintain, +/// checked from the kernel's own table rather than the driver's word for it. +/// +/// Only *registered* children are checked, not the whole tree. Firmware topology is +/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host +/// bridge's `bus_range`, because a bus-number range isn't an address window. +fn childrenContained() bool { + var buffer: [64]danos.DeviceDescriptor = undefined; + const n = @min(device_service.enumerate(&buffer), buffer.len); + + const bus_id = hpetDeviceId() orelse return false; + const p = buffer[@intCast(bus_id)]; + + var children: usize = 0; + for (buffer[0..n]) |d| { + if (d.parent != bus_id) continue; + children += 1; + for (0..d.resource_count) |i| { + const r = d.resources[i]; + var ok = false; + for (0..p.resource_count) |j| { + const pr = p.resources[j]; + if (pr.kind != r.kind) continue; + if (r.kind == @intFromEnum(danos.ResourceKind.irq)) { + if (pr.start == r.start) ok = true; + } else if (r.len != 0 and r.start >= pr.start and + r.start + r.len <= pr.start + pr.len) ok = true; + } + if (!ok) return false; + } + } + return children > 0; // busd must have published at least one +} + +/// Device id of the HPET (the bus busd claims), from the same table drivers see. +fn hpetDeviceId() ?u64 { + var buffer: [64]danos.DeviceDescriptor = undefined; + const n = @min(device_service.enumerate(&buffer), buffer.len); + for (buffer[0..n]) |d| { + if (d.class != @intFromEnum(danos.DeviceClass.timer)) continue; + if (d.parent != danos.no_parent) continue; // a comparator child, not the block + for (0..d.resource_count) |j| { + if (d.resources[j].kind == @intFromEnum(danos.ResourceKind.memory)) return d.id; + } + } + return null; +} + +/// IRQ teardown. When a driver exits, its bindings must be released: the line masked +/// (so a dead driver's device goes quiet instead of storming) and the slot cleared +/// (so an ISR never posts a notification into the endpoint that is about to be freed). +/// +/// This is the path `hpetd` never takes — it runs forever — so it gets its own test. +/// Two properties, both read back from the hardware rather than from our own state: +/// +/// 1. A bound GSI is routed and unmasked. +/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again. +/// +/// And one property that can only be checked from kernel state: a *different* owner's +/// binding on the same endpoint survives. Endpoints are shared (ipc_register hands out +/// references), so teardown keyed on the endpoint pointer rather than the owning task +/// would mask a live sibling driver's device line. +fn irqFreeTest() void { + log("DANOS-TEST-BEGIN: irqfree\n", .{}); + + const gsi = hpetGsi() orelse { + check("discovery recorded an IRQ resource for the HPET", false); + result(); + return; + }; + + const endpoint = ipcsync.createEndpoint() orelse { + check("allocated an endpoint", false); + result(); + return; + }; + + // Two owners, one shared endpoint. `other` binds a second line if the I/O APIC has + // one spare; if not, the sharing half of the test is skipped rather than faked. + const owner: u32 = 4242; + const other: u32 = 4343; + const spare: ?u32 = if (gsi + 1 < irq.maximum_gsi and architecture.irqOwnsGsi(gsi + 1)) gsi + 1 else null; + + { + const flags = sync.enter(); + defer sync.leave(flags); + irq.bind(gsi, endpoint, owner) catch {}; + if (spare) |s| irq.bind(s, endpoint, other) catch {}; + } + check("bound GSI is routed and unmasked", !entryMasked(gsi)); + if (spare) |s| check("second owner's GSI is routed and unmasked", !entryMasked(s)); + + { + const flags = sync.enter(); + defer sync.leave(flags); + irq.releaseOwner(owner); + } + check("exiting owner's line is masked again", entryMasked(gsi)); + if (spare) |s| { + check("a sibling owner's binding on the same endpoint survives", !entryMasked(s)); + const flags = sync.enter(); + defer sync.leave(flags); + irq.releaseOwner(other); + } + + result(); +} + +/// Is redirection entry `gsi` masked? (bit 16 of the low dword; entry index == GSI +/// because this I/O APIC's gsi_base is 0.) +fn entryMasked(gsi: u32) bool { + return architecture.irqRouteRaw(gsi) & (1 << 16) != 0; +} + /// The MMIO-grant teardown fix: a device-granted leaf must NOT be returned to the /// RAM allocator when its address space is destroyed. Map a real RAM frame as a /// device grant, tear the address space down, and confirm the frame is still held @@ -1091,20 +1298,20 @@ fn ioPassTest() void { log("DANOS-TEST-BEGIN: iopass\n", .{}); const base_free = pmm.stats().free_frames; - const aspace = arch.createAddressSpace() orelse { + const aspace = architecture.createAddressSpace() orelse { check("created a fresh address space", false); result(); return; }; const frame = pmm.alloc() orelse { - arch.destroyAddressSpace(aspace); + architecture.destroyAddressSpace(aspace); check("allocated a frame to grant", false); result(); return; }; // Map it the way mmio_map does (device grant), then tear the space down. - arch.mapUserDeviceInto(aspace, process.dev_arena_base, frame, danos.page_size); - arch.destroyAddressSpace(aspace); + architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, danos.page_size); + architecture.destroyAddressSpace(aspace); // The page tables were reclaimed; the device-granted frame must not have been. check("device-granted frame survived teardown (not reclaimed as RAM)", pmm.stats().free_frames == base_free - 1); @@ -1134,9 +1341,9 @@ fn faultNull() void { // 0 (otherwise it folds a null-pointer safety panic instead of doing the real // access). `allowzero` skips the same null check on the cast. The write then // hits the unmapped page 0 and takes a real hardware #PF. - var addr: u64 = 0; - addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr)); - const p: *allowzero volatile u64 = @ptrFromInt(addr); + var address: u64 = 0; + address = asm ("" : [ret] "=r" (-> u64) : [in] "0" (address)); + const p: *allowzero volatile u64 = @ptrFromInt(address); p.* = 1; } @@ -1144,15 +1351,15 @@ fn faultPageFault() void { log("DANOS-TEST-BEGIN: fault-pf\n", .{}); // Runtime address so the backend emits a register store (not a `mov moffs`, // which the self-hosted x86_64 backend can't encode). - var addr: u64 = 0xdeadbeef000; // well above all mapped RAM - const p: *volatile u64 = @ptrFromInt(addr); + var address: u64 = 0xdeadbeef000; // well above all mapped RAM + const p: *volatile u64 = @ptrFromInt(address); p.* = 1; - addr += 0; + address += 0; } fn faultDoubleFault() void { log("DANOS-TEST-BEGIN: fault-df\n", .{}); - arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF + architecture.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF // Point RSP at unmapped memory, then fault: the CPU can't push the fault // frame, which escalates to #DF — survivable only because #DF runs on IST1. var bad_sp: u64 = 0x5000000000; @@ -1172,9 +1379,9 @@ var ap_reached_fault: bool = false; /// core, then triggers the same double fault as `faultDoubleFault` — which is only /// survivable on IST1, so it exercises that core's own TSS. fn apDoubleFaultTask() void { - log("DANOS-AP: task running on core {d}, triggering #DF\n", .{sched.currentCpuIndex()}); + log("DANOS-AP: task running on core {d}, triggering #DF\n", .{scheduler.currentCpuIndex()}); @atomicStore(bool, &ap_reached_fault, true, .release); - arch.disableInterrupts(); + architecture.disableInterrupts(); var bad_sp: u64 = 0x5000000000; asm volatile ( \\mov %[sp], %%rsp @@ -1194,15 +1401,15 @@ fn apDoubleFaultTask() void { /// the fault was *contained* to the AP, not fatal to the system. fn faultApTest() void { log("DANOS-TEST-BEGIN: fault-ap-df\n", .{}); - if (!sched.spawnOn(apDoubleFaultTask, 6, 1)) { + if (!scheduler.spawnOn(apDoubleFaultTask, 6, 1)) { log("DANOS-AP: could not pin to core 1 (run with -smp) - FAIL\n", .{}); - arch.halt(); + architecture.halt(); } // Wait until the AP is about to fault, then keep running to prove containment. var spins: u64 = 0; while (!@atomicLoad(bool, &ap_reached_fault, .acquire) and spins < 5_000_000_000) spins +%= 1; var settle: u64 = 0; while (settle < 500_000_000) settle +%= 1; // let the AP take + report the fault - log("DANOS-BSP: core {d} still running after the AP fault (contained)\n", .{sched.currentCpuIndex()}); - arch.halt(); + log("DANOS-BSP: core {d} still running after the AP fault (contained)\n", .{scheduler.currentCpuIndex()}); + architecture.halt(); } diff --git a/src/config.zig b/src/parameters.zig similarity index 96% rename from src/config.zig rename to src/parameters.zig index 843967c..baf041c 100644 --- a/src/config.zig +++ b/src/parameters.zig @@ -13,11 +13,11 @@ /// stacks) are allocated at bring-up for cores that actually come online, so this /// ceiling is cheap. A machine with more logical CPUs has its surplus reported and /// left parked (see acpi `cpusDropped`). -pub const max_cpus = 128; +pub const maximum_cpus = 128; /// Maximum tasks (kernel threads) alive at once — the static task-table size. Each /// online core consumes one slot for its idle task, plus task 0 on the BSP. -pub const max_tasks = 16; +pub const maximum_tasks = 16; /// Each task's kernel stack (also each AP's bring-up stack), in bytes. pub const kernel_stack_size = 16 * 1024; diff --git a/src/root.zig b/src/root.zig index 2dffd40..d6d120a 100644 --- a/src/root.zig +++ b/src/root.zig @@ -6,10 +6,10 @@ const std = @import("std"); -/// Calling convention for the bootloader→kernel jump. Pinned to SysV so it does +/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does /// not depend on each binary's target default: the UEFI bootloader's C /// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is -/// SysV (first arg in RDI). Both reference this to agree on where `*BootInfo` +/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation` /// is passed. pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} }; @@ -47,23 +47,23 @@ pub const page_size = 4096; /// The kernel's virtual-memory layout (higher-half). The kernel is linked at /// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the -/// device MMIO windows) is also mapped at `physmap_base + phys`, so the kernel +/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel /// can reach any physical address by adding a constant. The low half is left /// entirely to user space. /// /// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half) /// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256]) -/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + phys +/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical /// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511]) pub const physmap_base: u64 = 0xFFFF_8800_0000_0000; pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000; -/// The kernel syscall numbers — the single source of truth shared by the kernel +/// The kernel system_call numbers — the single source of truth shared by the kernel /// dispatcher (src/kernel/process.zig) and the user runtime library, so the two /// can never drift. The set is deliberately microkernel-minimal: file/device I/O /// is not here — it lives in user-space servers reached through the IPC calls. /// The table grows one milestone at a time; see docs/syscall.md. -pub const Syscall = enum(u64) { +pub const SystemCall = enum(u64) { exit = 0, // exit(code): end the calling process yield = 1, // yield(): give up the rest of this quantum debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only) @@ -73,18 +73,26 @@ pub const Syscall = enum(u64) { create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint - ipc_call = 9, // ipc_call(h, msg, len, reply, cap) -> reply_len: send + block for reply - ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, recv, cap) -> recv_len (+badge in rdx) - dev_enumerate = 11, // dev_enumerate(buf, max) -> count: snapshot the device table - dev_claim = 12, // dev_claim(id) -> ok: take exclusive ownership of a device - mmio_map = 13, // mmio_map(id, res_idx) -> vaddr: map a claimed device's MMIO into this AS - irq_bind = 14, // irq_bind(id, res_idx, endpoint): deliver a device IRQ as an IPC notification - irq_ack = 15, // irq_ack(id, res_idx): re-arm a bound IRQ after servicing it + ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply + ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx) + device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table + device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device + mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS + irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification + irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it + device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed _, }; -/// A device class, mirroring src/device/device.zig's `DeviceClass` **in order** -/// (its `@intFromEnum` values cross the syscall boundary in `DeviceDesc.class`). +/// Set in the badge returned by `ipc_reply_wait` when what arrived is an +/// **asynchronous notification** (today: a device interrupt bound with `irq_bind`) +/// rather than a message from a client. There is no payload and no reply owed; the +/// low bits carry the source, a GSI. Shared so the kernel's ISR and the driver's +/// event loop can't disagree about which bit means "the hardware spoke". +pub const notify_badge_bit: u64 = 1 << 63; + +/// A device class, mirroring src/device/device-model.zig's `DeviceClass` **in order** +/// (its `@intFromEnum` values cross the system_call boundary in `DeviceDescriptor.class`). /// Keep the two in sync. pub const DeviceClass = enum(u32) { root, @@ -97,7 +105,7 @@ pub const DeviceClass = enum(u32) { unknown, }; -/// A resource kind, mirroring src/device/device.zig's `ResourceKind` in order. +/// A resource kind, mirroring src/device/device-model.zig's `ResourceKind` in order. pub const ResourceKind = enum(u32) { memory, io_port, @@ -106,23 +114,34 @@ pub const ResourceKind = enum(u32) { }; /// One device resource, as handed to a user-space driver (flat, extern). -pub const ResDesc = extern struct { +pub const ResourceDescriptor = extern struct { kind: u64, // a ResourceKind value start: u64, len: u64, }; -pub const max_dev_resources = 8; +pub const maximum_device_resources = 8; -/// A device, as snapshotted for user space by `dev_enumerate`. A driver scans +/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree. +pub const no_parent: u64 = ~@as(u64, 0); + +/// A device, as snapshotted for user space by `device_enumerate`. A driver scans /// these to find the hardware it owns, claims it, and maps its MMIO. -pub const DeviceDesc = extern struct { +/// +/// `parent` makes the table a tree rather than a list, which is what a **bus driver** +/// needs: it claims the bus, finds the devices below it, and publishes any it +/// discovers itself with `device_register`. A registered child's resources must lie +/// within its parent's (the kernel enforces this) — that containment is what makes +/// delegation safe, since a device descriptor is otherwise a licence to map physical +/// memory. +pub const DeviceDescriptor = extern struct { id: u64, + parent: u64, // a device id, or `no_parent` class: u64, // a DeviceClass value hid_len: u64, resource_count: u64, hid: [8]u8, - resources: [max_dev_resources]ResDesc, + resources: [maximum_device_resources]ResourceDescriptor, }; /// Well-known IPC service ids for the bootstrap name registry (create_endpoint + @@ -145,21 +164,21 @@ pub const prot_exec: u64 = 4; /// The bootloader may use the *constant* `physmap_base` to build those tables, /// but must not call this to dereference memory before its own CR3 is loaded — /// it runs under the firmware's identity map, where these addresses are unmapped. -pub inline fn physToVirt(phys: u64) u64 { - return phys + physmap_base; +pub inline fn physicalToVirtual(physical: u64) u64 { + return physical + physmap_base; } -/// Physmap virtual address -> physical. Inverse of `physToVirt`; for producing +/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing /// the physical address of something the kernel holds a physmap pointer to /// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location). -pub inline fn virtToPhys(virt: u64) u64 { - return virt - physmap_base; +pub inline fn virtualToPhysical(virtual: u64) u64 { + return virtual - physmap_base; } /// danos's own classification of a span of physical memory — deliberately not /// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its /// native memory description into these kinds, so the kernel never learns what -/// booted it. [[arch]] keeps the same discipline for CPU code. +/// booted it. [[architecture]] keeps the same discipline for CPU code. pub const MemoryKind = enum(u32) { /// Free RAM the kernel may allocate. Each boot path folds its own transient /// memory into this once it's genuinely free (e.g. the UEFI loader classifies @@ -174,7 +193,7 @@ pub const MemoryKind = enum(u32) { /// ACPI non-volatile storage: preserve across sleep, do not allocate. acpi_nvs, /// Not backed by RAM: memory-mapped device registers or a reserved - /// address-space window (e.g. PCIe config space). Kept distinct from + /// address-space window (e.g. PCIe configuration space). Kept distinct from /// `reserved` so RAM accounting doesn't count device address space. mmio, }; @@ -199,20 +218,20 @@ pub const MemoryMap = extern struct { /// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with /// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF -/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virt` is the higher-half link address; -/// `phys` is where the loader actually placed the segment (they differ once the +/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address; +/// `physical` is where the loader actually placed the segment (they differ once the /// kernel links high — the loader records the real load address here). pub const KernelSegment = extern struct { - virt: u64, - phys: u64, + virtual: u64, + physical: u64, pages: u64, flags: u32, _pad: u32 = 0, }; /// Handoff structure the bootloader fills in and passes to the kernel's -/// `_start` in RDI (the first argument under the SysV AMD64 C ABI). -pub const BootInfo = extern struct { +/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI). +pub const BootInformation = extern struct { framebuffer: Framebuffer, memory_map: MemoryMap, /// The kernel's own PT_LOAD segments (it has three: text, rodata, data). @@ -231,7 +250,7 @@ pub const BootInfo = extern struct { init_len: u64 = 0, /// The initrd image (a bundle of extra user binaries — the VFS server and /// device drivers), read off the boot volume into memory that survives the - /// handoff, same as `init` above. 0/0 = no initrd. See src/user/proto/initrd.zig. + /// handoff, same as `init` above. 0/0 = no initrd. See src/user/protocol/initrd.zig. initrd_base: u64 = 0, initrd_len: u64 = 0, }; diff --git a/src/user/proto/initrd.zig b/src/user/protocol/initrd.zig similarity index 100% rename from src/user/proto/initrd.zig rename to src/user/protocol/initrd.zig diff --git a/test/qemu_test.py b/test/qemu_test.py index a035db9..e4f4caa 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -188,11 +188,24 @@ CASES = [ {"name": "vfs", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, - # IO passthrough: a user-space HPET driver maps device MMIO into its own - # address space and reads the counter advancing. + # IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into + # its own address space, binds the device's interrupt to an IPC endpoint, and + # is woken by the hardware five times while blocked (never polling). {"name": "hpet", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Bus driver: a user process claims a device, enumerates its children from the + # hardware, and publishes each with dev_register — and the kernel refuses a child + # whose window escapes the parent's (else dev_register maps arbitrary memory). + {"name": "bus", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + # IRQ teardown: an exiting driver's line is masked and its slot cleared (so no + # ISR notifies a freed endpoint), and a sibling owner sharing that endpoint + # keeps its own binding. The path hpetd never takes, since it runs forever. + {"name": "irqfree", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # The MMIO-grant teardown fix: a device-granted frame must not be reclaimed # as RAM when its address space is destroyed. {"name": "iopass",