M11–M12: IRQ-as-IPC and bus drivers; expand names tree-wide
Two driver-model milestones plus a tree-wide naming pass. Suite 35/35 (QEMU) + host tests green. M11 — IRQ-as-IPC. A ring-3 driver now sleeps until its device interrupts it. New src/kernel/irq.zig: per-GSI endpoint bindings, comptime per-vector trampolines, dispatch = mask GSI -> LAPIC EOI -> notifyLocked, all under one lock region. irq_bind/irq_ack syscalls, gated by the device claim like mmio_map. interruptDispatch no longer EOIs — each handler owns its EOI, because a level line must be masked before it is acknowledged (irq_ack is the unmask). Bindings are keyed on the owning task and released on exit (a shared endpoint's siblings survive). hpetd rewritten interrupt-driven. Tests: hpet (rewritten, reads back the I/O APIC routing) and irqfree. M12 — bus drivers. DeviceDesc gains a parent, making the device table a tree. dev_register (device_register) lets a process publish children below a device it claimed; the kernel enforces resource containment (a child's resources must nest in its parent's), so a descriptor can't fabricate a window over kernel RAM. Descriptor copied in via copyFromUser (physmap walk — an unmapped user pointer fails the call instead of faulting the kernel). Per-parent child cap bounds table exhaustion. sbin/busd.zig is a worked bus driver. Test: bus. Naming — per docs/coding-standards.md: non-acronym abbreviations spelled out (message, descriptor, device_service, scheduler, runtime, physical, interpreter, ...); acronyms kept (IPC, MMIO, DMA, HCD, ...); files are kebab-case (ipc-synchronous.zig, device-service.zig, vfs-protocol.zig, ...). Exceptions: POSIX/C ABI names and Zig idioms (init/len/ptr) kept. Module collisions resolved by specific naming (config -> parameters, device.zig alias -> device_model). AML op/Op disambiguated: op = opcode, Op = operation; per-opcode parse handlers renamed opX -> parseX. New driver docs: drivers.md, driver-model.md (bus/class/HCD shapes + the proposed M13–M16 ABI), coding-standards.md.
This commit is contained in:
@@ -51,13 +51,13 @@ fn timestamp(b: *std.Build) []const u8 {
|
|||||||
/// Build one user-space binary the same way for every program (init, and later
|
/// Build one user-space binary the same way for every program (init, and later
|
||||||
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
||||||
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
||||||
/// can't reach), linked against the `rt` runtime library with the shared user
|
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||||
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||||
fn addUserBinary(
|
fn addUserBinary(
|
||||||
b: *std.Build,
|
b: *std.Build,
|
||||||
target: std.Build.ResolvedTarget,
|
target: std.Build.ResolvedTarget,
|
||||||
rt_mod: *std.Build.Module,
|
runtime_module: *std.Build.Module,
|
||||||
name: []const u8,
|
name: []const u8,
|
||||||
root: []const u8,
|
root: []const u8,
|
||||||
) *std.Build.Step.Compile {
|
) *std.Build.Step.Compile {
|
||||||
@@ -73,7 +73,7 @@ fn addUserBinary(
|
|||||||
.stack_check = false,
|
.stack_check = false,
|
||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "rt", .module = rt_mod },
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
@@ -91,74 +91,74 @@ pub fn build(b: *std.Build) void {
|
|||||||
const target = b.standardTargetOptions(.{});
|
const target = b.standardTargetOptions(.{});
|
||||||
const optimize = b.standardOptimizeOption(.{});
|
const optimize = b.standardOptimizeOption(.{});
|
||||||
|
|
||||||
// Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set,
|
// Shared handoff definitions (BootInformation, Framebuffer, ...). No target is set,
|
||||||
// so the module inherits the target of whichever binary imports it — the
|
// so the module inherits the target of whichever binary imports it — the
|
||||||
// freestanding kernel or the UEFI bootloader.
|
// freestanding kernel or the UEFI bootloader.
|
||||||
const mod = b.addModule("danos", .{
|
const danos_module = b.addModule("danos", .{
|
||||||
.root_source_file = b.path("src/root.zig"),
|
.root_source_file = b.path("src/root.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
// Kernel tunables (max_cpus, stack sizes, tick rate). A dependency-free module of
|
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||||
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||||
// in one place instead of scattered across the tree. See src/config.zig.
|
// in one place instead of scattered across the tree. See src/configuration.zig.
|
||||||
const config_mod = b.addModule("config", .{
|
const parameters_module = b.addModule("parameters", .{
|
||||||
.root_source_file = b.path("src/config.zig"),
|
.root_source_file = b.path("src/parameters.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
||||||
// The generic kernel imports this as "arch" and never names x86_64, so a new
|
// The generic kernel imports this as "architecture" and never names x86_64, so a new
|
||||||
// architecture is a matter of pointing this module at a different directory.
|
// architecture is a matter of pointing this module at a different directory.
|
||||||
const arch_mod = b.addModule("arch", .{
|
const architecture_module = b.addModule("architecture", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
|
.{ .name = "danos", .module = danos_module }, // paging uses the shared BootInformation/memory-map types
|
||||||
.{ .name = "config", .module = config_mod }, // max_cpus, ist_stack_size, timer_hz
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// CPU-exception stubs — real assembly, since they need cross-symbol
|
// CPU-exception stubs — real assembly, since they need cross-symbol
|
||||||
// jumps/calls that Zig inline asm can't express (see the file's header).
|
// jumps/calls that Zig inline asm can't express (see the file's header).
|
||||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
architecture_module.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
||||||
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
||||||
// inline asm (it runs relocated to a low page, not at its link address).
|
// inline asm (it runs relocated to a low page, not at its link address).
|
||||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s"));
|
architecture_module.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s"));
|
||||||
|
|
||||||
// Firmware-agnostic device discovery. The generic kernel imports this as
|
// Firmware-agnostic device discovery. The generic kernel imports this as
|
||||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||||
// arch module applies to CPU code. The backend is selected at runtime from
|
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||||
// the boot handoff (see src/device/platform.zig).
|
// the boot handoff (see src/device/platform.zig).
|
||||||
const platform_mod = b.addModule("platform", .{
|
const platform_module = b.addModule("platform", .{
|
||||||
.root_source_file = b.path("src/device/platform.zig"),
|
.root_source_file = b.path("src/device/platform.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP)
|
.{ .name = "danos", .module = danos_module }, // BootInformation (carries the ACPI RSDP)
|
||||||
.{ .name = "config", .module = config_mod }, // max_cpus (the discovery pool)
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
// The user-space runtime library (a nascent libc): syscall wrappers, the
|
// The user-space runtime library (a nascent libc): system_call wrappers, the
|
||||||
// C-convention heap, IPC helpers, the process start shim. Compiled into every
|
// C-convention heap, IPC helpers, the process start shim. Compiled into every
|
||||||
// user binary (see addUserBinary), so it inherits each exe's `.large` code
|
// user binary (see addUserBinary), so it inherits each exe's `.large` code
|
||||||
// model — do NOT set a target/code_model here. It imports `danos` for the
|
// model — do NOT set a target/code_model here. It imports `danos` for the
|
||||||
// shared Syscall numbers.
|
// shared SystemCall numbers.
|
||||||
const rt_mod = b.addModule("rt", .{
|
const runtime_module = b.addModule("runtime", .{
|
||||||
.root_source_file = b.path("lib/rt.zig"),
|
.root_source_file = b.path("lib/runtime.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
.{ .name = "danos", .module = danos_module },
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
// The initrd container format, shared by the kernel (unpacks it) and the
|
// The initrd container format, shared by the kernel (unpacks it) and the
|
||||||
// build-time packer tools/mkinitrd.zig (produces it). No dependencies.
|
// build-time packer tools/mkinitrd.zig (produces it). No dependencies.
|
||||||
const initrd_mod = b.addModule("initrd", .{
|
const initrd_module = b.addModule("initrd", .{
|
||||||
.root_source_file = b.path("src/user/proto/initrd.zig"),
|
.root_source_file = b.path("src/user/protocol/initrd.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
// Compile-time config the kernel reads as `@import("build_options")`. The
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
const build_options = b.addOptions();
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
const build_options_mod = build_options.createModule();
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -177,18 +177,18 @@ pub fn build(b: *std.Build) void {
|
|||||||
.target = kernel_target,
|
.target = kernel_target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
.{ .name = "danos", .module = danos_module },
|
||||||
.{ .name = "arch", .module = arch_mod },
|
.{ .name = "architecture", .module = architecture_module },
|
||||||
.{ .name = "platform", .module = platform_mod },
|
.{ .name = "platform", .module = platform_module },
|
||||||
.{ .name = "config", .module = config_mod },
|
.{ .name = "parameters", .module = parameters_module },
|
||||||
.{ .name = "build_options", .module = build_options_mod },
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
.{ .name = "initrd", .module = initrd_mod },
|
.{ .name = "initrd", .module = initrd_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
@@ -208,18 +208,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
|
|
||||||
// --- /sbin/init: the first user-space program ---
|
// --- /sbin/init: the first user-space program ---
|
||||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||||
// linked into the kernel's user region against the `rt` runtime library, and
|
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||||
// started in ring 3 by the kernel's user-ELF loader.
|
// started in ring 3 by the kernel's user-ELF loader.
|
||||||
const init_exe = addUserBinary(b, kernel_target, rt_mod, "init", "sbin/init.zig");
|
const init_exe = addUserBinary(b, kernel_target, runtime_module, "init", "sbin/init.zig");
|
||||||
b.installArtifact(init_exe);
|
b.installArtifact(init_exe);
|
||||||
|
|
||||||
// --- initrd: a bundle of extra user binaries (VFS server + drivers) ---
|
// --- initrd: a bundle of extra user binaries (VFS server + drivers) ---
|
||||||
// Each is built by the same user-binary recipe, then packed into one image by
|
// Each is built by the same user-binary recipe, then packed into one image by
|
||||||
// the host-side mkinitrd tool. The bootloader ferries the image to the kernel,
|
// the host-side mkinitrd tool. The bootloader ferries the image to the kernel,
|
||||||
// which unpacks it and spawns each program (src/user/proto/initrd.zig).
|
// which unpacks it and spawns each program (src/user/protocol/initrd.zig).
|
||||||
const vfs_exe = addUserBinary(b, kernel_target, rt_mod, "vfs", "sbin/vfs.zig");
|
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, "vfs", "sbin/vfs.zig");
|
||||||
const vfstest_exe = addUserBinary(b, kernel_target, rt_mod, "vfstest", "sbin/vfstest.zig");
|
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, "vfs-test", "sbin/vfs-test.zig");
|
||||||
const hpetd_exe = addUserBinary(b, kernel_target, rt_mod, "hpetd", "sbin/hpetd.zig");
|
const hpetd_exe = addUserBinary(b, kernel_target, runtime_module, "hpetd", "sbin/hpetd.zig");
|
||||||
|
const busd_exe = addUserBinary(b, kernel_target, runtime_module, "busd", "sbin/busd.zig");
|
||||||
|
|
||||||
// Pack the user binaries into the initrd image with the host-side Python tool
|
// Pack the user binaries into the initrd image with the host-side Python tool
|
||||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
@@ -229,10 +230,12 @@ pub fn build(b: *std.Build) void {
|
|||||||
const initrd_img = mk_run.addOutputFileArg("initrd.img");
|
const initrd_img = mk_run.addOutputFileArg("initrd.img");
|
||||||
mk_run.addArg("vfs");
|
mk_run.addArg("vfs");
|
||||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||||
mk_run.addArg("vfstest");
|
mk_run.addArg("vfs-test");
|
||||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||||
mk_run.addArg("hpetd");
|
mk_run.addArg("hpetd");
|
||||||
mk_run.addFileArg(hpetd_exe.getEmittedBin());
|
mk_run.addFileArg(hpetd_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("busd");
|
||||||
|
mk_run.addFileArg(busd_exe.getEmittedBin());
|
||||||
|
|
||||||
// Install the image to zig-out/bin (so the QEMU test harness picks it up like
|
// Install the image to zig-out/bin (so the QEMU test harness picks it up like
|
||||||
// the other binaries). The run-x86-64 ESP install is added below.
|
// the other binaries). The run-x86-64 ESP install is added below.
|
||||||
@@ -252,7 +255,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
}),
|
}),
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
.{ .name = "danos", .module = danos_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
@@ -261,14 +264,14 @@ pub fn build(b: *std.Build) void {
|
|||||||
|
|
||||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
// Firmware lives in different places per OS/distro, so probe the known
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
// layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||||
const ovmf_code = b.option(
|
const ovmf_code = b.option(
|
||||||
[]const u8,
|
[]const u8,
|
||||||
"ovmf-code",
|
"ovmf-code",
|
||||||
"Path to the OVMF_CODE firmware image",
|
"Path to the OVMF_CODE firmware image",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||||
@@ -280,7 +283,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
"ovmf-vars",
|
"ovmf-vars",
|
||||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||||
|
|||||||
+37
-7
@@ -35,9 +35,20 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||||
blocking (sleep, wait queues) — the leap to a running system.
|
blocking (sleep, wait queues) — the leap to a running system.
|
||||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||||
message-passing channels — the backbone the microkernel's isolated servers will
|
message-passing channels, then synchronous call/reply between *processes* over
|
||||||
talk over.
|
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||||
12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||||
|
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||||
|
deliberately tiny.
|
||||||
|
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||||
|
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||||
|
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||||
|
unmask.
|
||||||
|
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||||
|
real driver stacks factor into three shapes, how families share code, and the
|
||||||
|
proposed ABI for the three primitives still missing (capability passing, DMA +
|
||||||
|
memory barriers, MSI).
|
||||||
|
15. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
@@ -68,6 +79,10 @@ Cutting across all of these:
|
|||||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||||
right choice depends on whether danos is chasing real-time or resilience.
|
right choice depends on whether danos is chasing real-time or resilience.
|
||||||
|
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||||
|
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||||
|
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||||
|
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||||
@@ -94,19 +109,34 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
|||||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||||
([halting.md](halting.md)).
|
([halting.md](halting.md)).
|
||||||
|
|
||||||
|
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||||
|
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||||
|
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||||
|
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||||
|
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||||
|
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||||
|
|
||||||
## Source map
|
## Source map
|
||||||
|
|
||||||
| Area | Code |
|
| Area | Code |
|
||||||
|------|------|
|
|------|------|
|
||||||
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||||
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
||||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
|
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, `Syscall`, ABI) | `src/root.zig` |
|
||||||
| Physical frame allocator | `src/kernel/pmm.zig` |
|
| Physical frame allocator | `src/kernel/pmm.zig` |
|
||||||
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
||||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
|
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/scheduler.zig` |
|
||||||
| IPC channels (message passing) | `src/kernel/ipc.zig` |
|
| Big kernel lock + interrupt-safe critical sections | `src/kernel/sync.zig` |
|
||||||
|
| IPC channels between kernel threads (message passing) | `src/kernel/ipc.zig` |
|
||||||
|
| IPC endpoints: cross-address-space call/reply, handles, notifications | `src/kernel/ipc-synchronous.zig` |
|
||||||
|
| User processes: ELF loading, address spaces, the syscall table | `src/kernel/process.zig` |
|
||||||
|
| Device tree + claim capability + `device_register` containment | `src/kernel/device-service.zig` |
|
||||||
|
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `src/kernel/irq.zig` |
|
||||||
|
| Hardware discovery (ACPI/device tree) behind one neutral device model | `src/device/` |
|
||||||
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
||||||
| In-kernel test cases | `src/kernel/tests.zig` |
|
| In-kernel test cases | `src/kernel/tests.zig` |
|
||||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
||||||
|
| User runtime library (`rt`): syscalls, heap, stdio, IPC, device access | `lib/` |
|
||||||
|
| User-space programs shipped in the initrd (`init`, `vfs`, `hpetd` leaf driver, `busd` bus driver) | `sbin/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
@@ -0,0 +1,137 @@
|
|||||||
|
# Coding standards
|
||||||
|
|
||||||
|
Conventions for danos source. The overriding one, from which most of the rest follows:
|
||||||
|
|
||||||
|
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||||
|
> abbreviation is an acronym.**
|
||||||
|
|
||||||
|
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||||
|
is a Zig idiom — see the exceptions). `device_service`, not `device_service`. `scheduler`, not
|
||||||
|
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||||
|
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||||
|
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||||
|
that trade is not close.
|
||||||
|
|
||||||
|
## The rule, precisely
|
||||||
|
|
||||||
|
**Acronyms and initialisms stay.** They *are* the full name — expanding them would make
|
||||||
|
the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`,
|
||||||
|
`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`,
|
||||||
|
`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`,
|
||||||
|
`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`,
|
||||||
|
`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention
|
||||||
|
demands: `Hal` the type, `hal` the variable, `mapMmio` the function.
|
||||||
|
|
||||||
|
**Everything else is spelled out.** If it's a word with letters removed, restore them:
|
||||||
|
|
||||||
|
| Abbreviation | Full |
|
||||||
|
|---|---|
|
||||||
|
| `proto` | `protocol` |
|
||||||
|
| `msg` | `message` |
|
||||||
|
| `desc` | `descriptor` |
|
||||||
|
| `res` | `resource` |
|
||||||
|
| `recv` | `receive` |
|
||||||
|
| `buf` | `buffer` |
|
||||||
|
| `cur` | `current` |
|
||||||
|
| `src` / `dst` | `source` / `destination` |
|
||||||
|
| `idx` | `index` |
|
||||||
|
| `addr` | `address` |
|
||||||
|
| `reg` | `register` |
|
||||||
|
| `prev` | `previous` |
|
||||||
|
| `cfg` / `config` | `configuration` |
|
||||||
|
| `arch` | `architecture` |
|
||||||
|
| `sched` | `scheduler` |
|
||||||
|
| `dev` | `device` |
|
||||||
|
| `sys` / `syscall` | `system` / `system_call` |
|
||||||
|
| `info` | `information` |
|
||||||
|
| `dt` | `device_tree` |
|
||||||
|
| `ep` | `endpoint` |
|
||||||
|
| `rt` | `runtime` |
|
||||||
|
| `func` | `function` |
|
||||||
|
| `phys` / `virt` | `physical` / `virtual` |
|
||||||
|
| `wq` | `wait_queue` |
|
||||||
|
|
||||||
|
This list is illustrative, not exhaustive. The rule is the rule; when you meet a new
|
||||||
|
abbreviation, expand it.
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
Three, and only three.
|
||||||
|
|
||||||
|
1. **Foreign ABI names are spelled exactly as the ABI spells them.** A function that
|
||||||
|
*is* the C or POSIX interface keeps its name: `fopen`, `fwrite`, `fread`, `malloc`,
|
||||||
|
`calloc`, `realloc`, `free`, `memcpy`, `mmap`, `munmap`, `open`, `read`, `write`,
|
||||||
|
`close`, `lseek`, `stat`, `errno`. We don't get to rename `fwrite` to
|
||||||
|
`fileWrite` — it wouldn't be `fwrite` any more. This also covers the syscall
|
||||||
|
*wrappers* that exist to match those names. It does **not** license inventing new
|
||||||
|
abbreviated names in that style.
|
||||||
|
|
||||||
|
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||||
|
not ours, and are left alone:
|
||||||
|
- **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a
|
||||||
|
shortening of "initialize".
|
||||||
|
- **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own
|
||||||
|
structs use bare `len`/`ptr` fields to mirror them, so a reader carries one
|
||||||
|
mental model. (Compounds still expand: a field is `message_len`, not
|
||||||
|
`message_length` — `len` is kept, `msg` is not.)
|
||||||
|
- The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are
|
||||||
|
Zig's spelling.
|
||||||
|
|
||||||
|
The rule governs the names *we* coin.
|
||||||
|
|
||||||
|
3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may
|
||||||
|
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||||
|
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||||
|
|
||||||
|
4. **Established Unix filesystem and program conventions.** Top-level directories keep
|
||||||
|
their conventional names — `src`, `lib`, `sbin`, `bin`, `docs` — as do daemon
|
||||||
|
programs by their `d` suffix (`hpetd`, `busd`, following `sshd`/`httpd`). These are
|
||||||
|
names a Unix reader already knows; expanding them fights the convention rather than
|
||||||
|
serving it.
|
||||||
|
|
||||||
|
## A note on collisions
|
||||||
|
|
||||||
|
Two identifiers can legitimately expand to the same word. When they do, keep both
|
||||||
|
meaningful by renaming one to its *specific* identity rather than the generic
|
||||||
|
expansion. Two cases resolved this way:
|
||||||
|
|
||||||
|
- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would
|
||||||
|
collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module
|
||||||
|
became **`parameters`**, which is what it holds.
|
||||||
|
- The kernel `device.zig` module would collide with `dev` (a device value) at
|
||||||
|
`device`. The module alias became **`device_model`**, which is what it is — the
|
||||||
|
device data model (`Device`, `DeviceTree`, `ResourceKind`).
|
||||||
|
- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a
|
||||||
|
`Namespace` **instance**. Resolved by dropping the module alias entirely — the two
|
||||||
|
types it provided are imported directly (`const Node = @import("namespace.zig").Node;`)
|
||||||
|
— which frees `namespace` for the instance.
|
||||||
|
|
||||||
|
A related case is one abbreviation with two meanings. In the AML code, `op` means
|
||||||
|
**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` /
|
||||||
|
`LogicOperation` means **operation** — distinguished by case. The per-opcode parser
|
||||||
|
handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the
|
||||||
|
opcode's structure, which says what they do without overloading "op".
|
||||||
|
|
||||||
|
## Case and file names
|
||||||
|
|
||||||
|
Within those spelling rules, follow Zig's own conventions:
|
||||||
|
|
||||||
|
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||||
|
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||||
|
- **Variables, fields, constants** — `snake_case`: `message_length`, `device_service`,
|
||||||
|
`notify_badge_bit`.
|
||||||
|
|
||||||
|
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||||
|
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `device-service.zig`. A
|
||||||
|
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||||
|
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||||
|
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||||
|
|
||||||
|
## Why acronyms are the line
|
||||||
|
|
||||||
|
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||||
|
input output" in code — that expansion is what the acronym *is for*. But `msg` is just
|
||||||
|
`message` with three letters stolen, and stealing them buys nothing a reader wants. The
|
||||||
|
test for "is this an abbreviation I must expand" is simply: *is there a longer word this
|
||||||
|
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||||
|
phrase, leave it.
|
||||||
+32
-11
@@ -89,7 +89,6 @@ if (state.vector < 32) {
|
|||||||
on_fault(state); // exception: report and halt (never returns)
|
on_fault(state); // exception: report and halt (never returns)
|
||||||
} else if (handlers[state.vector]) |handler| {
|
} else if (handlers[state.vector]) |handler| {
|
||||||
handler(); // device: run the registered handler
|
handler(); // device: run the registered handler
|
||||||
apic.eoi(); // ...acknowledge the LAPIC
|
|
||||||
}
|
}
|
||||||
// else: spurious/unhandled — deliberately no EOI
|
// else: spurious/unhandled — deliberately no EOI
|
||||||
```
|
```
|
||||||
@@ -100,10 +99,23 @@ Two things make device interrupts *return* where exceptions don't:
|
|||||||
flows back to `isr_common`, which restores every register it saved and executes
|
flows back to `isr_common`, which restores every register it saved and executes
|
||||||
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
||||||
saves *all* the general registers.)
|
saves *all* the general registers.)
|
||||||
2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss
|
2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss
|
||||||
this and the LAPIC thinks we're still busy and never delivers the next
|
this and the LAPIC thinks we're still busy and never delivers the next
|
||||||
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
||||||
|
|
||||||
|
**Each handler issues its own EOI**, rather than the dispatcher doing it around the
|
||||||
|
call. That looks like a needless devolution while the timer is the only device, and
|
||||||
|
`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early,
|
||||||
|
because the tick hook is the scheduler, which may switch tasks and not return
|
||||||
|
promptly — the LAPIC mustn't wait on it).
|
||||||
|
|
||||||
|
It stops looking needless with the second device. A *routed* interrupt — one arriving
|
||||||
|
through the I/O APIC from a real device line — must be **masked before it is
|
||||||
|
acknowledged**, because a level-triggered line is still asserted at EOI time and would
|
||||||
|
redeliver instantly, forever. Only the handler knows which discipline its source
|
||||||
|
needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the
|
||||||
|
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||||
|
|
||||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||||
a handler must not use them; ours don't.)
|
a handler must not use them; ours don't.)
|
||||||
@@ -131,14 +143,23 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the
|
|||||||
count would stay put and the test would fail. That it advances — while the CPU was
|
count would stay put and the test would fail. That it advances — while the CPU was
|
||||||
spinning in unrelated code — is the whole mechanism working end to end.
|
spinning in unrelated code — is the whole mechanism working end to end.
|
||||||
|
|
||||||
|
## Since (done elsewhere)
|
||||||
|
|
||||||
|
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||||
|
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||||
|
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||||
|
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||||
|
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||||
|
user drivers — see [paging.md](paging.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read
|
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and ring 3
|
||||||
scancodes from the PS/2 controller — the first *input* device.
|
has no port I/O yet, so the first *input* device is blocked on either an I/O
|
||||||
- **`sleep()` / timeouts** built on the calibrated clock (the monotonic
|
permission bitmap or `io_in`/`io_out` syscalls ([drivers.md](drivers.md)).
|
||||||
`uptimeMs()` is in place).
|
- **MSI/MSI-X**: per-device vectors, edge-triggered and unshared, which retire the
|
||||||
- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like
|
I/O APIC's mask/ack cycle and its 24-GSI ceiling.
|
||||||
the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO
|
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||||
marked uncacheable (via the page's cache bits or an MTRR).
|
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||||
- **Preemption**: once there are tasks, the timer handler is where the scheduler
|
|
||||||
decides to switch — the reason a *returning* interrupt matters.
|
|
||||||
|
|||||||
@@ -0,0 +1,305 @@
|
|||||||
|
# The driver model: buses, classes, and host controllers
|
||||||
|
|
||||||
|
[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its
|
||||||
|
registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is
|
||||||
|
not enough for a disk, a keyboard, or a network card, because those hang off a
|
||||||
|
*controller*, on a *bus*, speaking a *protocol*, and no single process should have to
|
||||||
|
know all three.
|
||||||
|
|
||||||
|
Real driver stacks factor into three shapes. This document is about what each one is,
|
||||||
|
what the kernel must give it, how they share code — and precisely which primitive each
|
||||||
|
is still blocked on.
|
||||||
|
|
||||||
|
## Three shapes
|
||||||
|
|
||||||
|
| Shape | Owns | Reaches hardware by | Talks to |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language |
|
||||||
|
| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC |
|
||||||
|
| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC |
|
||||||
|
|
||||||
|
The last row is the surprising one and the whole point. A USB keyboard driver touches
|
||||||
|
no registers, takes no interrupts, and maps no memory. It sends HID protocol messages
|
||||||
|
to whatever published the device, and it works identically whether the controller
|
||||||
|
below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that
|
||||||
|
outlive the hardware they were written for.
|
||||||
|
|
||||||
|
In practice **HCD and bus driver are usually the same process**. An xHCI driver is a
|
||||||
|
host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA
|
||||||
|
rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting
|
||||||
|
them is a fiction; what matters is that both *roles* have kernel support, because a
|
||||||
|
plain bus driver with no controller — a USB hub — is also a real thing.
|
||||||
|
|
||||||
|
## The device table is the spine
|
||||||
|
|
||||||
|
danos already has the right central structure. `src/kernel/device-service.zig` holds a table of
|
||||||
|
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||||
|
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||||
|
|
||||||
|
Three invariants make it a capability system rather than a directory:
|
||||||
|
|
||||||
|
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||||
|
`mmio_map`, `irq_bind`, `device_register` — checks `device_service.ownerOf(id) == me`.
|
||||||
|
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||||
|
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||||
|
`device_register` cannot be a free-for-all.
|
||||||
|
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||||
|
resource of the same kind on its parent (`device_service.contains`). A bus driver can only
|
||||||
|
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||||
|
syscall named "map any physical page you like."
|
||||||
|
|
||||||
|
Containment is transitive by construction: a grandchild is contained in its child,
|
||||||
|
which is contained in the bus. Nothing can be laundered through a chain.
|
||||||
|
|
||||||
|
Note that firmware topology does **not** obey containment, and isn't asked to — a PCI
|
||||||
|
function's BAR is not inside its host bridge's `bus_range`, because a bus-number range
|
||||||
|
is not an address window. Discovery is trusted; user space is not.
|
||||||
|
|
||||||
|
### What a bus driver looks like
|
||||||
|
|
||||||
|
`sbin/busd.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||||
|
its "devices" are the block's comparators:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
_ = dev.claim(bus.id); // 1. own the bus
|
||||||
|
const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware
|
||||||
|
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||||
|
|
||||||
|
for (0..n) |i| { // 3. publish each child
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory,
|
||||||
|
.start = bus_mmio.start + 0x100 + 0x20 * i,
|
||||||
|
.len = 0x20 };
|
||||||
|
_ = dev.register(bus.id, &child).?; // kernel checks containment
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||||
|
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||||
|
whose window escapes the bus is refused — `busd` asserts that, and the `bus` test
|
||||||
|
asserts the kernel's table upholds it.
|
||||||
|
|
||||||
|
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||||
|
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||||
|
|
||||||
|
## Families: sharing code between drivers
|
||||||
|
|
||||||
|
A "family" is two modules, not one:
|
||||||
|
|
||||||
|
- **A logic module** — the parts of the bus that every driver on it re-derives. Config
|
||||||
|
space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub
|
||||||
|
protocol for USB.
|
||||||
|
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||||
|
*whatever* published its device. This is the part that makes class drivers portable.
|
||||||
|
|
||||||
|
danos already has one of each: `lib/device.zig` is a logic module,
|
||||||
|
[`lib/vfs-protocol.zig`](lib/vfs-protocol.zig) is a protocol module shared by `sbin/vfs.zig`
|
||||||
|
and its clients. The pattern generalises directly:
|
||||||
|
|
||||||
|
```
|
||||||
|
lib/
|
||||||
|
rt.zig module "rt" — syscalls, heap, ipc, dev, stdio
|
||||||
|
mmio.zig module "mmio" — volatile register access + barriers [M14]
|
||||||
|
bus/
|
||||||
|
pci.zig module "pci" — ECAM, BAR decode, capability walk
|
||||||
|
usb.zig module "usb" — descriptors, control transfers, hubs
|
||||||
|
proto/
|
||||||
|
vfs.zig module "proto.vfs" (today: lib/vfs-protocol.zig)
|
||||||
|
block.zig module "proto.block"
|
||||||
|
hid.zig module "proto.hid"
|
||||||
|
|
||||||
|
sbin/
|
||||||
|
xhcid.zig HCD + bus driver imports rt, pci, usb, mmio
|
||||||
|
usbhid.zig class driver imports rt, usb, proto.hid
|
||||||
|
blockd.zig class driver imports rt, proto.block
|
||||||
|
```
|
||||||
|
|
||||||
|
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||||
|
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||||
|
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||||
|
everything else.
|
||||||
|
|
||||||
|
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||||
|
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||||
|
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||||
|
|
||||||
|
## What exists today
|
||||||
|
|
||||||
|
- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device
|
||||||
|
grants, `device_grant` teardown.
|
||||||
|
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||||
|
EOI; `irq_ack` is the unmask.
|
||||||
|
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||||
|
|
||||||
|
So: **bus drivers work now.** HCDs and class drivers do not. Here is exactly why, and
|
||||||
|
exactly what would fix it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Proposed ABI
|
||||||
|
|
||||||
|
## M13 — capability passing, for class drivers
|
||||||
|
|
||||||
|
**The blocker.** A class driver has to reach *its* device. Today the only way to find
|
||||||
|
an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`,
|
||||||
|
where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot
|
||||||
|
mint one endpoint per USB device that way, and there is no way for a bus driver to
|
||||||
|
*hand* a class driver an endpoint. M7 deferred this deliberately.
|
||||||
|
|
||||||
|
**The fix.** Let a message carry one handle. Sender names a handle in its own table;
|
||||||
|
the kernel installs the endpoint into the receiver's table (bumping `refcount`) and
|
||||||
|
tells the receiver the index it landed at.
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len
|
||||||
|
ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap)
|
||||||
|
-> recv_len (rax), badge (rdx), received_cap (r8)
|
||||||
|
```
|
||||||
|
|
||||||
|
`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred
|
||||||
|
endpoint was installed at in the receiver's table, or `no_cap`.
|
||||||
|
|
||||||
|
- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`,
|
||||||
|
leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`;
|
||||||
|
this needs a third (`setSyscallResult3`).
|
||||||
|
- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is
|
||||||
|
not delivered** — a half-delivered capability is worse than a failed send.
|
||||||
|
- `closeHandles` already drops references on exit, so the lifetime story is unchanged.
|
||||||
|
|
||||||
|
That single primitive gives you the standard `open` pattern:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
// class driver // bus driver
|
||||||
|
const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...);
|
||||||
|
const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||||
|
.{ .op = .open, .id = dev_id }); // reply with it as send_cap
|
||||||
|
// now dev_ep is a private channel to that one device
|
||||||
|
```
|
||||||
|
|
||||||
|
## M14 — DMA memory and the memory-ordering contract, for HCDs
|
||||||
|
|
||||||
|
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
|
||||||
|
device can read, which means memory that is (a) physically contiguous, (b) at a
|
||||||
|
physical address the driver knows, (c) of the right cacheability, and (d) pinned.
|
||||||
|
[`sysMmap`](src/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
|
||||||
|
once per page, maps writeback-cached, and never reveals a physical address.
|
||||||
|
|
||||||
|
**The fix.**
|
||||||
|
|
||||||
|
```
|
||||||
|
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||||
|
dma_free(vaddr, len) -> 0
|
||||||
|
|
||||||
|
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||||
|
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||||
|
dma_below_4g (4) for devices with 32-bit DMA addressing
|
||||||
|
```
|
||||||
|
|
||||||
|
Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the
|
||||||
|
mapping, and the physical address is stable. It needs one thing the kernel lacks —
|
||||||
|
`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time
|
||||||
|
with no adjacency guarantee.
|
||||||
|
|
||||||
|
**The memory-ordering contract.** danos has, at the time of writing, **zero memory
|
||||||
|
barriers anywhere in the tree.** That is currently correct-by-accident and won't
|
||||||
|
survive the first DMA driver, or the first ARM boot.
|
||||||
|
|
||||||
|
`volatile` is not a barrier. In Zig it means: don't elide this access, and don't
|
||||||
|
reorder it against *other volatile* accesses. It says nothing about your *ordinary*
|
||||||
|
stores — the descriptor you just filled in normal WB memory — which LLVM may freely
|
||||||
|
sink past a volatile MMIO write. The canonical bug:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||||
|
```
|
||||||
|
|
||||||
|
So the rules, which belong in `lib/mmio.zig` and behind `arch`:
|
||||||
|
|
||||||
|
| Situation | Required |
|
||||||
|
|---|---|
|
||||||
|
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||||
|
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||||
|
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||||
|
| MMIO write that must complete before the next read | `mb()` |
|
||||||
|
|
||||||
|
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||||
|
sprinkling of `asm volatile`:
|
||||||
|
|
||||||
|
| | x86_64 | aarch64 |
|
||||||
|
|---|---|---|
|
||||||
|
| `mb()` | `mfence` | `dsb sy` |
|
||||||
|
| `rmb()` | `lfence` | `dsb ld` |
|
||||||
|
| `wmb()` | `sfence` | `dsb st` |
|
||||||
|
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||||
|
|
||||||
|
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||||
|
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||||
|
condition. Build the abstraction while there is one caller to fix.
|
||||||
|
|
||||||
|
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||||
|
barrier, or per-arch inline asm — which is what `lib/mmio.zig` should hide.)
|
||||||
|
|
||||||
|
## M15 — interrupts for PCI devices
|
||||||
|
|
||||||
|
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||||
|
[`addBars`](src/device/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||||
|
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpetd` only works because the
|
||||||
|
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||||
|
device has.
|
||||||
|
|
||||||
|
**The fix, in two halves.**
|
||||||
|
|
||||||
|
*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it
|
||||||
|
as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**,
|
||||||
|
and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the
|
||||||
|
line must be polled on each interrupt — the reason everyone left INTx behind.
|
||||||
|
|
||||||
|
*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no
|
||||||
|
mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver
|
||||||
|
the (address, data) pair to program into its own MSI capability:
|
||||||
|
|
||||||
|
```
|
||||||
|
msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 }
|
||||||
|
```
|
||||||
|
|
||||||
|
The driver writes those into config space itself — which means it needs config space,
|
||||||
|
which means **discovery should give each `pci_device` a `.memory` resource for its
|
||||||
|
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||||
|
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||||
|
|
||||||
|
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpetd` can never
|
||||||
|
exercise this path. The first MSI driver will be the first PCI driver.
|
||||||
|
|
||||||
|
## M16 — the IOMMU, and the honest caveat
|
||||||
|
|
||||||
|
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||||
|
A driver that can program a bus-mastering engine can make that device write to any
|
||||||
|
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||||
|
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||||
|
on any DMA-capable device is equivalent to granting ring 0.**
|
||||||
|
|
||||||
|
This does not make the model useless — it's the same position Linux is in with the
|
||||||
|
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||||
|
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||||
|
gap should be named rather than implied.
|
||||||
|
|
||||||
|
## Ordering
|
||||||
|
|
||||||
|
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||||
|
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||||
|
could write a real one against `busd`'s comparators tomorrow.
|
||||||
|
|
||||||
|
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||||
|
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||||
|
from hand-rolling `*volatile` and getting ARM wrong.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||||
|
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||||
|
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||||
|
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||||
+319
@@ -0,0 +1,319 @@
|
|||||||
|
# Writing a driver
|
||||||
|
|
||||||
|
In a monolithic kernel a driver is a function call away from everything: it runs in
|
||||||
|
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||||
|
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||||
|
can crash without taking the kernel with it, and — the point of this document — it
|
||||||
|
can be restarted ([resilience](resilience.md)).
|
||||||
|
|
||||||
|
That leaves three questions the kernel has to answer, because a process can't answer
|
||||||
|
them for itself:
|
||||||
|
|
||||||
|
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||||
|
([discovery](discovery.md), [acpi](acpi.md)).
|
||||||
|
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||||
|
device's physical MMIO window into your address space, and from then on it's plain
|
||||||
|
memory. No syscall per register access.
|
||||||
|
3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered
|
||||||
|
to you as an IPC notification. You block; the hardware wakes you.
|
||||||
|
|
||||||
|
A driver is, in one sentence, *a process that sleeps until its device has something to
|
||||||
|
say.*
|
||||||
|
|
||||||
|
## The capability: claim before touch
|
||||||
|
|
||||||
|
The five driver syscalls (`src/root.zig`, dispatched in `src/kernel/process.zig`):
|
||||||
|
|
||||||
|
| # | Call | Meaning |
|
||||||
|
|---|------|---------|
|
||||||
|
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||||
|
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||||
|
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||||
|
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||||
|
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||||
|
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||||
|
|
||||||
|
Notice that **nothing takes a physical address or an interrupt number.** Every call
|
||||||
|
names a device by id and a resource by index. That indirection is the entire security
|
||||||
|
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||||
|
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||||
|
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||||
|
the same check at the top of `sysMmioMap`):
|
||||||
|
|
||||||
|
- `device_service.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||||
|
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||||
|
for `irq_bind`
|
||||||
|
|
||||||
|
The claim is the capability. Everything else follows from it.
|
||||||
|
|
||||||
|
## Registers: `mmio_map`
|
||||||
|
|
||||||
|
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||||
|
with `present | user | writable | nx | pcd | pwt`
|
||||||
|
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||||
|
|
||||||
|
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||||
|
would return a stale value and a write might never leave the CPU.
|
||||||
|
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||||
|
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||||
|
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||||
|
the frame allocator as if they were free RAM. The `iopass` test guards it.
|
||||||
|
|
||||||
|
Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages
|
||||||
|
never widen an existing mapping.
|
||||||
|
|
||||||
|
Then you just… use it:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const base = dev.mmioMap(dev_id, mmio_res) orelse return;
|
||||||
|
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
||||||
|
const now = counter.*; // a load, straight to the hardware. no kernel involved.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Interrupts: the cycle, and why it has that shape
|
||||||
|
|
||||||
|
An interrupt handler in a microkernel has a problem. The code that knows how to quiet
|
||||||
|
the device is in ring 3, in another address space, and it will not run for
|
||||||
|
microseconds or milliseconds — after a context switch, when the scheduler gets to it.
|
||||||
|
But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until
|
||||||
|
the device is quieted. EOI a still-asserted line and the I/O APIC redelivers
|
||||||
|
immediately. Forever. The driver never gets to run at all.
|
||||||
|
|
||||||
|
The way out is to mask the line before acknowledging it:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU
|
||||||
|
irqEoi() // now safe to tell the LAPIC we're done
|
||||||
|
notifyFromIsr() // wake the driver — it runs much later
|
||||||
|
|
||||||
|
driver replyWait() -> badge with the notify bit set
|
||||||
|
<clear the device's status register> // NOW the line deasserts
|
||||||
|
irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the
|
||||||
|
interrupt fires exactly once, ever; call it before the device is quiet and you get an
|
||||||
|
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||||
|
syscalls and not one.
|
||||||
|
|
||||||
|
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||||
|
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||||
|
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||||
|
handler knows which discipline its source needs.
|
||||||
|
|
||||||
|
### The driver side is an event loop, not a callback
|
||||||
|
|
||||||
|
`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by
|
||||||
|
the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one
|
||||||
|
single-threaded loop over both of its event sources:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
while (true) {
|
||||||
|
const r = ipc.replyWait(endpoint, reply, &recv);
|
||||||
|
if (r.isNotification()) { // r.source() is the GSI
|
||||||
|
service_device(); // clear the status register
|
||||||
|
_ = dev.irqAck(id, irq_res); // re-arm
|
||||||
|
} else {
|
||||||
|
handle_client_request(recv[0..r.len]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
No reentrancy, no "what am I allowed to call from an interrupt handler", no shared
|
||||||
|
state between ISR and task context. The interrupt is just a message.
|
||||||
|
|
||||||
|
Two properties worth knowing:
|
||||||
|
|
||||||
|
- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in
|
||||||
|
an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody
|
||||||
|
waiting, but the badge is already on the endpoint's notify ring. The next
|
||||||
|
`replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO).
|
||||||
|
- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on
|
||||||
|
overflow. That's correct: an IRQ notification is a *level* ("the device wants
|
||||||
|
attention"), not a tally. Re-read the device's status register; never assume one
|
||||||
|
notification means exactly one event. Because the ISR masks the line until you
|
||||||
|
`irq_ack`, at most one badge per GSI can be outstanding — so the ring can only
|
||||||
|
overflow if you bind more than eight GSIs to a single endpoint. Don't.
|
||||||
|
|
||||||
|
## A whole driver
|
||||||
|
|
||||||
|
`sbin/hpetd.zig` is ~150 lines and does all of it. The shape:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||||
|
// class=timer with memory + irq
|
||||||
|
_ = dev.claim(hpet.dev_id); // the capability
|
||||||
|
const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?;
|
||||||
|
const endpoint = ipc.createEndpoint().?;
|
||||||
|
|
||||||
|
// program the hardware over the mapping we were just handed
|
||||||
|
reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator
|
||||||
|
reg(base, 0x010).* |= 1; // ENABLE
|
||||||
|
|
||||||
|
_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint);
|
||||||
|
|
||||||
|
while (...) {
|
||||||
|
const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling.
|
||||||
|
if (r.badge & notify_bit == 0) continue;
|
||||||
|
reg(base, 0x020).* = 1; // clear status -> deassert
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm
|
||||||
|
_ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
||||||
|
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
||||||
|
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||||
|
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||||
|
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||||
|
mask/ack cycle above is exercised for real rather than being decoration on an
|
||||||
|
edge-triggered line that would have been fine without it.
|
||||||
|
|
||||||
|
One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**.
|
||||||
|
Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in
|
||||||
|
the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the
|
||||||
|
mask, and records one concrete GSI as an `irq` resource. The driver then programs
|
||||||
|
`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one
|
||||||
|
it recorded. Hardware that describes itself at runtime still has to fit through a
|
||||||
|
static capability.
|
||||||
|
|
||||||
|
## Publishing children: `device_register`
|
||||||
|
|
||||||
|
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||||
|
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||||
|
That's `device_register`, and it makes the device table a tree rather than a list
|
||||||
|
(`DeviceDesc.parent`).
|
||||||
|
|
||||||
|
```zig
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||||
|
const child_id = dev.register(bus_id, &child).?;
|
||||||
|
```
|
||||||
|
|
||||||
|
The child is left **unclaimed**, which is the whole point: another process claims it and
|
||||||
|
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||||
|
|
||||||
|
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||||
|
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||||
|
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||||
|
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||||
|
bus driver may only ever subdivide what it already owns.
|
||||||
|
|
||||||
|
A device with **no resources** is legal and common. A USB device is reached through its
|
||||||
|
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||||
|
|
||||||
|
See [`sbin/busd.zig`](../sbin/busd.zig) for a complete one, and
|
||||||
|
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||||
|
controller drivers fit together.
|
||||||
|
|
||||||
|
## What the kernel does not do for you
|
||||||
|
|
||||||
|
- **It does not quiet your device.** That's the whole reason `irq_ack` exists.
|
||||||
|
- **It does not know your registers.** `mmio_map` hands you a base address; every
|
||||||
|
offset in this document came from the HPET spec, not from danos.
|
||||||
|
- **It does not serialise your driver.** Two clients calling one driver endpoint are
|
||||||
|
serialised by `replyWait`, but nothing stops your driver from being preempted.
|
||||||
|
|
||||||
|
## Limits, today
|
||||||
|
|
||||||
|
Worth knowing before you write the second driver:
|
||||||
|
|
||||||
|
- **Ring 3 has no port I/O.** The TSS I/O permission bitmap is absent
|
||||||
|
(`tss.zig`: `iomap_base = @sizeOf(Tss)`), and IOPL is never raised, so `in`/`out`
|
||||||
|
from a driver is a #GP. That rules out a user-space 16550 UART (`0x3F8`), PS/2
|
||||||
|
(`0x60`/`0x64`), and legacy PCI config (`0xCF8`/`0xCFC`). Everything must be MMIO.
|
||||||
|
`io_port` resources are recorded by discovery and then ignored.
|
||||||
|
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||||
|
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||||
|
than a page, but its *mapping* can't.
|
||||||
|
- **No DMA memory.** `mmap` gives you writeback-cached, non-contiguous pages and never
|
||||||
|
tells you their physical address, so you cannot build a descriptor ring. Any driver
|
||||||
|
for a bus-mastering device is blocked on this.
|
||||||
|
- **No memory barriers.** There are none in the tree, and `volatile` is not one — it
|
||||||
|
won't stop the compiler sinking an ordinary store (your DMA descriptor) past a
|
||||||
|
volatile MMIO store (your doorbell). On x86 you mostly get away with it; on ARM you
|
||||||
|
will not. See [driver-model.md](driver-model.md#m14).
|
||||||
|
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||||
|
that device write to *any* physical address — page tables don't sit between a device
|
||||||
|
and RAM; an IOMMU does. Until VT-d/DMAR is programmed, `device_claim` on a DMA-capable
|
||||||
|
device is effectively equivalent to granting ring 0. This is the largest gap between
|
||||||
|
the design's promise and what it delivers.
|
||||||
|
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||||
|
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||||
|
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||||
|
- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override
|
||||||
|
says active-low needs that threaded through from discovery.
|
||||||
|
- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and
|
||||||
|
by a single I/O APIC.
|
||||||
|
- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops
|
||||||
|
on overflow. With one GSI per endpoint that's unreachable — the line is masked from
|
||||||
|
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||||
|
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||||
|
left to ack it.
|
||||||
|
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||||
|
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||||
|
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||||
|
[not built yet](resilience.md).
|
||||||
|
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||||
|
line on exit, but the claim is never released — restart is
|
||||||
|
[not built](resilience.md).
|
||||||
|
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||||
|
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||||
|
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||||
|
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
|
||||||
|
back. See the note at the top of `src/kernel/irq.zig`.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
The `hpet` test spawns `hpetd` from the initrd and watches the serial log. The driver
|
||||||
|
prints `hpetd: ok` only after being woken five times, and its loop's only exit is
|
||||||
|
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||||
|
|
||||||
|
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||||
|
APIC redirection entry back and asserts the line really is routed to a device vector,
|
||||||
|
really is level-triggered, and really was left unmasked by the driver's final
|
||||||
|
`irq_ack`.
|
||||||
|
|
||||||
|
```
|
||||||
|
$ python3 test/qemu_test.py hpet irqfree iopass
|
||||||
|
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
```
|
||||||
|
|
||||||
|
Two companions cover what `hpetd` can't, because it never exits:
|
||||||
|
|
||||||
|
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||||
|
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||||
|
is not. That second half is why bindings are keyed on the owning *task* and not on
|
||||||
|
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
||||||
|
this endpoint" would silently mask a live driver's device.
|
||||||
|
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||||
|
space never returns MMIO frames to the RAM pool.
|
||||||
|
|
||||||
|
## What's next (not done here)
|
||||||
|
|
||||||
|
The big ones — capability passing (class drivers), DMA + barriers and MSI (host
|
||||||
|
controller drivers), and the IOMMU — have proposed signatures in
|
||||||
|
[driver-model.md](driver-model.md). Smaller items:
|
||||||
|
|
||||||
|
- **Port I/O grants**, so a PS/2 or 16550 driver is possible: either a per-device TSS
|
||||||
|
I/O permission bitmap swapped on context switch, or `io_in`/`io_out` syscalls gated
|
||||||
|
by the same claim. The legacy devices that need it are all low-rate, so the syscall
|
||||||
|
is likely fast enough.
|
||||||
|
- **Releasing a claim.** There is no `dev_release`, and `device_service` never drops a claim on
|
||||||
|
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||||
|
which blocks restart.
|
||||||
|
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||||
|
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||||
|
table.
|
||||||
|
- **Restart.** A driver that dies should release its claim, have its device quiesced,
|
||||||
|
and be respawned by a supervisor. Some pieces (`releaseIrqs`, `device_grant`
|
||||||
|
teardown, the claim table) exist; the policy doesn't.
|
||||||
|
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||||
|
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||||
|
a woken driver waits for the next scheduling point.
|
||||||
+55
-12
@@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what
|
|||||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||||
concern, not an afterthought.
|
concern, not an afterthought.
|
||||||
|
|
||||||
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
There are two layers, built a milestone apart:
|
||||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
|
||||||
scheduler's [wait queues](scheduling.md).
|
- **`src/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||||
|
described below. The primitive, and where the blocking discipline was worked out.
|
||||||
|
- **`src/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||||
|
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||||
|
second half of this document.
|
||||||
|
|
||||||
## The channel
|
## The channel
|
||||||
|
|
||||||
|
The first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
||||||
|
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||||
|
scheduler's [wait queues](scheduling.md).
|
||||||
|
|
||||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||||
ring buffer, a count, and two wait queues:
|
ring buffer, a count, and two wait queues:
|
||||||
|
|
||||||
@@ -42,16 +50,51 @@ full and empty over and over, so both the blocking-send and blocking-recv paths
|
|||||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||||
|
|
||||||
|
## Endpoints: call/reply across address spaces
|
||||||
|
|
||||||
|
A channel connects two kernel threads sharing one address space. Real servers are
|
||||||
|
*processes*, so the payload has to cross an address-space boundary. That's
|
||||||
|
`src/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||||
|
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||||
|
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||||
|
bounce buffer).
|
||||||
|
|
||||||
|
Two syscalls carry it:
|
||||||
|
|
||||||
|
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||||
|
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||||
|
any), then block for the next request. One syscall, because a server's steady state
|
||||||
|
is *always* "finish the last one, wait for the next".
|
||||||
|
|
||||||
|
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||||
|
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||||
|
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||||
|
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||||
|
client calls `ipc_lookup(service_id)`.
|
||||||
|
|
||||||
|
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||||
|
the message: the caller's task id.
|
||||||
|
|
||||||
|
### Interrupts are messages too
|
||||||
|
|
||||||
|
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||||
|
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||||
|
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||||
|
client wants something" from "the hardware wants something". Notifications sit in a
|
||||||
|
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||||
|
elsewhere is not lost.
|
||||||
|
|
||||||
|
This is what makes a user-space driver possible at all, and it's the subject of
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **Across address spaces.** Today both endpoints are kernel threads sharing the
|
|
||||||
kernel's memory, so the message is copied within one address space. When user
|
|
||||||
mode arrives, the same channel carries messages between *isolated* processes,
|
|
||||||
copying the payload across the boundary — which is where IPC earns its place as
|
|
||||||
the microkernel's backbone.
|
|
||||||
- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply)
|
|
||||||
on top of channels, the shape most driver/service calls take.
|
|
||||||
- **Interrupts as messages.** A hardware interrupt delivered to the driver task
|
|
||||||
that owns the device, as an IPC message.
|
|
||||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||||
low-priority server doesn't suffer unbounded priority inversion.
|
low-priority server doesn't suffer unbounded priority inversion.
|
||||||
|
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||||
|
every capability is either well-known (the registry) or inherited — there's no way
|
||||||
|
to delegate one.
|
||||||
|
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||||
|
shape (logging, notifications between servers).
|
||||||
|
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||||
|
lock; a bulk transfer wants shared pages, not a copy.
|
||||||
|
|||||||
-32
@@ -1,32 +0,0 @@
|
|||||||
//! User-space device access: enumerate the kernel's device table, claim a
|
|
||||||
//! device, and map its MMIO. A driver uses these to find and take ownership of
|
|
||||||
//! its hardware; the claim is the capability the kernel checks before mapping.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const sc = @import("syscall.zig");
|
|
||||||
|
|
||||||
pub const DeviceDesc = danos.DeviceDesc;
|
|
||||||
pub const ResDesc = danos.ResDesc;
|
|
||||||
pub const DeviceClass = danos.DeviceClass;
|
|
||||||
pub const ResourceKind = danos.ResourceKind;
|
|
||||||
|
|
||||||
inline fn failed(r: usize) bool {
|
|
||||||
return r > ~@as(usize, 0) - 4095;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Copy up to `buf.len` device descriptors into `buf`; returns the total count.
|
|
||||||
pub fn enumerate(buf: []DeviceDesc) usize {
|
|
||||||
return sc.syscall2(.dev_enumerate, @intFromPtr(buf.ptr), buf.len);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
|
||||||
pub fn claim(id: u64) bool {
|
|
||||||
return !failed(sc.syscall1(.dev_claim, id));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map resource `res_idx` (which must be an MMIO window) of claimed device
|
|
||||||
/// `dev_id` into this address space; returns the register base virtual address.
|
|
||||||
pub fn mmioMap(dev_id: u64, res_idx: u64) ?usize {
|
|
||||||
const r = sc.syscall2(.mmio_map, dev_id, res_idx);
|
|
||||||
return if (failed(r)) null else r;
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,67 @@
|
|||||||
|
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||||
|
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||||
|
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||||
|
//! mapping registers or routing an IRQ.
|
||||||
|
|
||||||
|
const danos = @import("danos");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
pub const DeviceDescriptor = danos.DeviceDescriptor;
|
||||||
|
pub const ResourceDescriptor = danos.ResourceDescriptor;
|
||||||
|
pub const DeviceClass = danos.DeviceClass;
|
||||||
|
pub const ResourceKind = danos.ResourceKind;
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||||
|
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||||
|
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||||
|
pub fn claim(id: u64) bool {
|
||||||
|
return !failed(sc.systemCall1(.device_claim, id));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||||
|
/// `device_id` into this address space; returns the register base virtual address.
|
||||||
|
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||||
|
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||||
|
pub const no_parent = danos.no_parent;
|
||||||
|
|
||||||
|
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||||
|
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||||
|
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||||
|
///
|
||||||
|
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||||
|
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||||
|
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||||
|
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||||
|
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||||
|
/// through its controller, not by MMIO.
|
||||||
|
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||||
|
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||||
|
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||||
|
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||||
|
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||||
|
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||||
|
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||||
|
/// status register holds its line asserted) — the kernel left the line masked
|
||||||
|
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||||
|
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||||
|
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||||
|
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||||
|
}
|
||||||
+27
-27
@@ -5,16 +5,16 @@
|
|||||||
//! The algorithm is a straight port of the kernel's first-fit free list
|
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||||
//! (src/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
//! (src/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||||
//! split on allocation and coalesced with neighbours on free. The only thing
|
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||||
//! that changes on this side of the syscall boundary is where memory comes from
|
//! that changes on this side of the system_call boundary is where memory comes from
|
||||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||||
//! itself, and the kernel picks the base address.
|
//! itself, and the kernel picks the base address.
|
||||||
//!
|
//!
|
||||||
//! Single-threaded and 16-byte max alignment, exactly like the kernel heap; a
|
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||||
//! lock and larger alignments come when user programs gain threads.
|
//! lock and larger alignments come when user programs gain threads.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const sys = @import("sys.zig");
|
const system = @import("system.zig");
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
const page_size = danos.page_size;
|
||||||
|
|
||||||
@@ -26,8 +26,8 @@ const Block = extern struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
const header_size = @sizeOf(Block); // 16
|
const header_size = @sizeOf(Block); // 16
|
||||||
const min_block = header_size + 16; // smallest block worth splitting off
|
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||||
/// Grow granularity: one `mmap` per 64 KiB amortises the syscall.
|
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||||
const chunk = 64 * 1024;
|
const chunk = 64 * 1024;
|
||||||
|
|
||||||
var free_list: ?*Block = null;
|
var free_list: ?*Block = null;
|
||||||
@@ -44,10 +44,10 @@ fn payloadOf(block: *Block) [*]u8 {
|
|||||||
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||||
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||||
fn grow(min_bytes: usize) bool {
|
fn grow(minimum_bytes: usize) bool {
|
||||||
const bytes = alignUp(@max(min_bytes, chunk), page_size);
|
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||||
const ret = sys.mmap(bytes, sys.PROT_READ | sys.PROT_WRITE);
|
const ret = system.mmap(bytes, system.PROT_READ | system.PROT_WRITE);
|
||||||
if (sys.mmapFailed(ret)) return false;
|
if (system.mmapFailed(ret)) return false;
|
||||||
|
|
||||||
const block: *Block = @ptrFromInt(ret);
|
const block: *Block = @ptrFromInt(ret);
|
||||||
block.size = bytes;
|
block.size = bytes;
|
||||||
@@ -58,25 +58,25 @@ fn grow(min_bytes: usize) bool {
|
|||||||
/// Insert a block into the address-ordered free list, coalescing with the
|
/// Insert a block into the address-ordered free list, coalescing with the
|
||||||
/// physically adjacent free blocks on either side.
|
/// physically adjacent free blocks on either side.
|
||||||
fn insertFree(block: *Block) void {
|
fn insertFree(block: *Block) void {
|
||||||
var prev: ?*Block = null;
|
var previous: ?*Block = null;
|
||||||
var cur = free_list;
|
var current = free_list;
|
||||||
while (cur) |c| : (cur = c.next) {
|
while (current) |c| : (current = c.next) {
|
||||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||||
prev = c;
|
previous = c;
|
||||||
}
|
}
|
||||||
|
|
||||||
block.next = cur;
|
block.next = current;
|
||||||
if (prev) |p| p.next = block else free_list = block;
|
if (previous) |p| p.next = block else free_list = block;
|
||||||
|
|
||||||
// Merge forward into `cur` if they're contiguous.
|
// Merge forward into `current` if they're contiguous.
|
||||||
if (cur) |c| {
|
if (current) |c| {
|
||||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||||
block.size += c.size;
|
block.size += c.size;
|
||||||
block.next = c.next;
|
block.next = c.next;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Merge `prev` forward into `block` if they're contiguous.
|
// Merge `previous` forward into `block` if they're contiguous.
|
||||||
if (prev) |p| {
|
if (previous) |p| {
|
||||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||||
p.size += block.size;
|
p.size += block.size;
|
||||||
p.next = block.next;
|
p.next = block.next;
|
||||||
@@ -90,24 +90,24 @@ fn rawAlloc(len: usize) ?[*]u8 {
|
|||||||
|
|
||||||
var attempts: u32 = 0;
|
var attempts: u32 = 0;
|
||||||
while (attempts < 2) : (attempts += 1) {
|
while (attempts < 2) : (attempts += 1) {
|
||||||
var prev: ?*Block = null;
|
var previous: ?*Block = null;
|
||||||
var cur = free_list;
|
var current = free_list;
|
||||||
while (cur) |block| : ({
|
while (current) |block| : ({
|
||||||
prev = block;
|
previous = block;
|
||||||
cur = block.next;
|
current = block.next;
|
||||||
}) {
|
}) {
|
||||||
if (block.size < need) continue;
|
if (block.size < need) continue;
|
||||||
|
|
||||||
if (block.size >= need + min_block) {
|
if (block.size >= need + minimum_block) {
|
||||||
// Split: carve `need` off the front, leave the rest free.
|
// Split: carve `need` off the front, leave the rest free.
|
||||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||||
rest.size = block.size - need;
|
rest.size = block.size - need;
|
||||||
rest.next = block.next;
|
rest.next = block.next;
|
||||||
if (prev) |p| p.next = rest else free_list = rest;
|
if (previous) |p| p.next = rest else free_list = rest;
|
||||||
block.size = need;
|
block.size = need;
|
||||||
} else {
|
} else {
|
||||||
// Take the whole block.
|
// Take the whole block.
|
||||||
if (prev) |p| p.next = block.next else free_list = block.next;
|
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||||
}
|
}
|
||||||
return payloadOf(block);
|
return payloadOf(block);
|
||||||
}
|
}
|
||||||
|
|||||||
+30
-14
@@ -4,7 +4,7 @@
|
|||||||
//! added with the first server binary.
|
//! added with the first server binary.
|
||||||
|
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const sc = @import("syscall.zig");
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
/// A small-int handle into the calling process's handle table.
|
/// A small-int handle into the calling process's handle table.
|
||||||
pub const Handle = usize;
|
pub const Handle = usize;
|
||||||
@@ -18,61 +18,77 @@ pub const Message = extern struct {
|
|||||||
c: u64 = 0,
|
c: u64 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Whether a syscall return value is a wrapped -errno (lands in the top page).
|
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||||
inline fn failed(r: usize) bool {
|
inline fn failed(r: usize) bool {
|
||||||
return r > ~@as(usize, 0) - 4095;
|
return r > ~@as(usize, 0) - 4095;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a new endpoint owned by this process; returns its handle.
|
/// Create a new endpoint owned by this process; returns its handle.
|
||||||
pub fn createEndpoint() ?Handle {
|
pub fn createEndpoint() ?Handle {
|
||||||
const r = sc.syscall0(.create_endpoint);
|
const r = sc.systemCall0(.create_endpoint);
|
||||||
return if (failed(r)) null else r;
|
return if (failed(r)) null else r;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||||
pub fn register(id: danos.ServiceId, h: Handle) bool {
|
pub fn register(id: danos.ServiceId, h: Handle) bool {
|
||||||
return !failed(sc.syscall2(.ipc_register, @intFromEnum(id), h));
|
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||||
/// process.
|
/// process.
|
||||||
pub fn lookup(id: danos.ServiceId) ?Handle {
|
pub fn lookup(id: danos.ServiceId) ?Handle {
|
||||||
const r = sc.syscall1(.ipc_lookup, @intFromEnum(id));
|
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||||
return if (failed(r)) null else r;
|
return if (failed(r)) null else r;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const CallError = error{Failed};
|
pub const CallError = error{Failed};
|
||||||
|
|
||||||
/// Send `msg` to endpoint `h` and block until the server replies into `reply`.
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||||
/// Returns the reply length.
|
/// Returns the reply length.
|
||||||
pub fn call(h: Handle, msg: []const u8, reply: []u8) CallError!usize {
|
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||||
const r = sc.syscall5(.ipc_call, h, @intFromPtr(msg.ptr), msg.len, @intFromPtr(reply.ptr), reply.len);
|
const r = sc.systemCall5(.ipc_call, h, @intFromPtr(message.ptr), message.len, @intFromPtr(reply.ptr), reply.len);
|
||||||
return if (failed(r)) error.Failed else r;
|
return if (failed(r)) error.Failed else r;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||||
|
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||||
|
/// GSI. See `isNotification`.
|
||||||
|
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||||
|
|
||||||
/// The result of a `replyWait`: the request length and the sender's badge (a
|
/// The result of a `replyWait`: the request length and the sender's badge (a
|
||||||
/// task id, or an IRQ notification if the high bit is set).
|
/// task id, or an IRQ notification if the high bit is set).
|
||||||
pub const Received = struct {
|
pub const Received = struct {
|
||||||
len: usize,
|
len: usize,
|
||||||
badge: u64,
|
badge: u64,
|
||||||
|
|
||||||
|
/// True if this wake-up was a device interrupt, not a client request. A driver's
|
||||||
|
/// event loop branches on this; there is no reply owed on the notification path.
|
||||||
|
pub fn isNotification(self: Received) bool {
|
||||||
|
return self.badge & notify_badge_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The interrupt source (a GSI), meaningful only when `isNotification`.
|
||||||
|
pub fn source(self: Received) u64 {
|
||||||
|
return self.badge & ~notify_badge_bit;
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if
|
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if
|
||||||
/// any), then block until the next request arrives in `recv`. Returns its length
|
/// any), then block until the next request arrives in `receive`. Returns its length
|
||||||
/// and the sender badge. This syscall returns two values — the length in rax and
|
/// and the sender badge. This system_call returns two values — the length in rax and
|
||||||
/// the badge in rdx — so it needs a hand-written stub: rdx is a read-write
|
/// the badge in rdx — so it needs a hand-written stub: rdx is a read-write
|
||||||
/// operand (input = reply length, arg #3; output = badge).
|
/// operand (input = reply length, arg #3; output = badge).
|
||||||
pub fn replyWait(h: Handle, reply: []const u8, recv: []u8) Received {
|
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8) Received {
|
||||||
var rax: usize = undefined;
|
var rax: usize = undefined;
|
||||||
var rdx: usize = reply.len; // in: reply_len (arg #3 -> rdx); out: badge
|
var rdx: usize = reply.len; // in: reply_len (arg #3 -> rdx); out: badge
|
||||||
asm volatile ("syscall"
|
asm volatile ("syscall"
|
||||||
: [rax] "={rax}" (rax),
|
: [rax] "={rax}" (rax),
|
||||||
[rdx] "+{rdx}" (rdx),
|
[rdx] "+{rdx}" (rdx),
|
||||||
: [n] "{rax}" (@intFromEnum(danos.Syscall.ipc_reply_wait)),
|
: [n] "{rax}" (@intFromEnum(danos.SystemCall.ipc_reply_wait)),
|
||||||
[a0] "{rdi}" (h),
|
[a0] "{rdi}" (h),
|
||||||
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||||
[a3] "{r10}" (@intFromPtr(recv.ptr)),
|
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||||
[a4] "{r8}" (recv.len),
|
[a4] "{r8}" (receive.len),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
return .{ .len = rax, .badge = rdx };
|
return .{ .len = rax, .badge = rdx };
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,29 +1,29 @@
|
|||||||
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||||
//! and later the VFS server + device drivers) imports this as `@import("rt")`:
|
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||||
//! syscall wrappers, the C-convention heap, IPC helpers, and the process start
|
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||||
//! freestanding target), so all user programs share one implementation.
|
//! freestanding target), so all user programs share one implementation.
|
||||||
//!
|
//!
|
||||||
//! A user binary needs three lines:
|
//! A user binary needs three lines:
|
||||||
//! const rt = @import("rt");
|
//! const runtime = @import("runtime");
|
||||||
//! pub const panic = rt.panic;
|
//! pub const panic = runtime.panic;
|
||||||
//! comptime { _ = &rt.start._start; } // pull the entry shim in
|
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||||
//! and a `pub fn main() void`.
|
//! and a `pub fn main() void`.
|
||||||
|
|
||||||
pub const sys = @import("sys.zig");
|
pub const system = @import("system.zig");
|
||||||
pub const heap = @import("heap.zig");
|
pub const heap = @import("heap.zig");
|
||||||
pub const ipc = @import("ipc.zig");
|
pub const ipc = @import("ipc.zig");
|
||||||
pub const start = @import("start.zig");
|
pub const start = @import("start.zig");
|
||||||
/// The VFS wire protocol (shared with the VFS server).
|
/// The VFS wire protocol (shared with the VFS server).
|
||||||
pub const vfsproto = @import("vfs_proto.zig");
|
pub const vfs_protocol = @import("vfs-protocol.zig");
|
||||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||||
pub const unistd = @import("unistd.zig");
|
pub const unistd = @import("unistd.zig");
|
||||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||||
pub const stdio = @import("stdio.zig");
|
pub const stdio = @import("stdio.zig");
|
||||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||||
pub const dev = @import("dev.zig");
|
pub const device = @import("device.zig");
|
||||||
|
|
||||||
/// Re-exported so a user binary can `pub const panic = rt.panic;`.
|
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||||
pub const panic = start.panic;
|
pub const panic = start.panic;
|
||||||
|
|
||||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||||
+5
-5
@@ -1,11 +1,11 @@
|
|||||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||||
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||||
//! `comptime { _ = &rt.start._start; }`, so the whole runtime is linked in.
|
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const sys = @import("sys.zig");
|
const system = @import("system.zig");
|
||||||
|
|
||||||
/// The kernel enters at `_start` with rsp 16-aligned, but a SysV function expects
|
/// The kernel enters at `_start` with rsp 16-aligned, but a SystemV function expects
|
||||||
/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes
|
/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes
|
||||||
/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the
|
/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the
|
||||||
/// `ud2` is a safety net if `rt_start` ever returns.
|
/// `ud2` is a safety net if `rt_start` ever returns.
|
||||||
@@ -21,12 +21,12 @@ pub export fn _start() callconv(.naked) noreturn {
|
|||||||
export fn rt_start() callconv(.c) noreturn {
|
export fn rt_start() callconv(.c) noreturn {
|
||||||
const root = @import("root"); // the user binary's root source file
|
const root = @import("root"); // the user binary's root source file
|
||||||
root.main();
|
root.main();
|
||||||
sys.exit(0);
|
system.exit(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||||
pub const panic = std.debug.FullPanic(struct {
|
pub const panic = std.debug.FullPanic(struct {
|
||||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||||
sys.exit(127);
|
system.exit(127);
|
||||||
}
|
}
|
||||||
}.panic);
|
}.panic);
|
||||||
|
|||||||
+4
-4
@@ -8,7 +8,7 @@ const unistd = @import("unistd.zig");
|
|||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
|
|
||||||
pub const SEEK_SET = unistd.SEEK_SET;
|
pub const SEEK_SET = unistd.SEEK_SET;
|
||||||
pub const SEEK_CUR = unistd.SEEK_CUR;
|
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||||
pub const SEEK_END = unistd.SEEK_END;
|
pub const SEEK_END = unistd.SEEK_END;
|
||||||
|
|
||||||
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
||||||
@@ -47,10 +47,10 @@ pub fn fclose(f: *FILE) c_int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
||||||
pub fn fread(buf: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||||
const total = size * nmemb;
|
const total = size * nmemb;
|
||||||
if (total == 0) return 0;
|
if (total == 0) return 0;
|
||||||
const n = unistd.read(f.fd, buf[0..@min(buf.len, total)]);
|
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
||||||
if (n <= 0) {
|
if (n <= 0) {
|
||||||
f.eof = 1;
|
f.eof = 1;
|
||||||
return 0;
|
return 0;
|
||||||
@@ -76,7 +76,7 @@ pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn ftell(f: *FILE) i64 {
|
pub fn ftell(f: *FILE) i64 {
|
||||||
return unistd.lseek(f.fd, 0, unistd.SEEK_CUR);
|
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn rewind(f: *FILE) void {
|
pub fn rewind(f: *FILE) void {
|
||||||
|
|||||||
@@ -1,50 +1,50 @@
|
|||||||
//! Raw `syscall` instruction wrappers for user space — one per arity.
|
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||||
//!
|
//!
|
||||||
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||||
//! The `syscall` instruction itself clobbers rcx (it holds the return rip) and
|
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||||
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||||
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||||
//! instruction, so the kernel reads the 4th argument from r10.
|
//! instruction, so the kernel reads the 4th argument from r10.
|
||||||
|
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const Syscall = danos.Syscall;
|
const SystemCall = danos.SystemCall;
|
||||||
|
|
||||||
pub inline fn syscall0(n: Syscall) usize {
|
pub inline fn systemCall0(n: SystemCall) usize {
|
||||||
return asm volatile ("syscall"
|
return asm volatile ("syscall"
|
||||||
: [ret] "={rax}" (-> usize),
|
: [ret] "={rax}" (-> usize),
|
||||||
: [n] "{rax}" (@intFromEnum(n)),
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
}
|
}
|
||||||
|
|
||||||
pub inline fn syscall1(n: Syscall, a0: usize) usize {
|
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||||
return asm volatile ("syscall"
|
return asm volatile ("syscall"
|
||||||
: [ret] "={rax}" (-> usize),
|
: [ret] "={rax}" (-> usize),
|
||||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0),
|
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
}
|
}
|
||||||
|
|
||||||
pub inline fn syscall2(n: Syscall, a0: usize, a1: usize) usize {
|
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||||
return asm volatile ("syscall"
|
return asm volatile ("syscall"
|
||||||
: [ret] "={rax}" (-> usize),
|
: [ret] "={rax}" (-> usize),
|
||||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1),
|
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
}
|
}
|
||||||
|
|
||||||
pub inline fn syscall3(n: Syscall, a0: usize, a1: usize, a2: usize) usize {
|
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||||
return asm volatile ("syscall"
|
return asm volatile ("syscall"
|
||||||
: [ret] "={rax}" (-> usize),
|
: [ret] "={rax}" (-> usize),
|
||||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2),
|
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
}
|
}
|
||||||
|
|
||||||
pub inline fn syscall4(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||||
return asm volatile ("syscall"
|
return asm volatile ("syscall"
|
||||||
: [ret] "={rax}" (-> usize),
|
: [ret] "={rax}" (-> usize),
|
||||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3),
|
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
}
|
}
|
||||||
|
|
||||||
pub inline fn syscall5(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||||
return asm volatile ("syscall"
|
return asm volatile ("syscall"
|
||||||
: [ret] "={rax}" (-> usize),
|
: [ret] "={rax}" (-> usize),
|
||||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), [a4] "{r8}" (a4),
|
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), [a4] "{r8}" (a4),
|
||||||
@@ -1,9 +1,9 @@
|
|||||||
//! Typed syscall surface for user space — thin wrappers over the raw `syscall`
|
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||||
//! stubs, one per kernel call. Numbers come from `danos.Syscall`, the single
|
//! stubs, one per kernel call. Numbers come from `danos.SystemCall`, the single
|
||||||
//! source of truth shared with the kernel dispatcher.
|
//! source of truth shared with the kernel dispatcher.
|
||||||
|
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const sc = @import("syscall.zig");
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||||
@@ -13,23 +13,23 @@ pub const PROT_EXEC: usize = danos.prot_exec;
|
|||||||
|
|
||||||
/// Give up the rest of this quantum.
|
/// Give up the rest of this quantum.
|
||||||
pub fn yield() void {
|
pub fn yield() void {
|
||||||
_ = sc.syscall0(.yield);
|
_ = sc.systemCall0(.yield);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||||
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||||
pub fn write(msg: []const u8) usize {
|
pub fn write(message: []const u8) usize {
|
||||||
return sc.syscall2(.debug_write, @intFromPtr(msg.ptr), msg.len);
|
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Block the caller for `ms` milliseconds.
|
/// Block the caller for `ms` milliseconds.
|
||||||
pub fn sleep(ms: usize) void {
|
pub fn sleep(ms: usize) void {
|
||||||
_ = sc.syscall1(.sleep, ms);
|
_ = sc.systemCall1(.sleep, ms);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// End the process. Never returns.
|
/// End the process. Never returns.
|
||||||
pub fn exit(code: usize) noreturn {
|
pub fn exit(code: usize) noreturn {
|
||||||
_ = sc.syscall1(.exit, code);
|
_ = sc.systemCall1(.exit, code);
|
||||||
unreachable; // the kernel never returns from exit
|
unreachable; // the kernel never returns from exit
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -37,12 +37,12 @@ pub fn exit(code: usize) noreturn {
|
|||||||
/// memory and return the base virtual address. On failure returns a value in the
|
/// memory and return the base virtual address. On failure returns a value in the
|
||||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||||
pub fn mmap(len: usize, prot: usize) usize {
|
pub fn mmap(len: usize, prot: usize) usize {
|
||||||
return sc.syscall2(.mmap, len, prot);
|
return sc.systemCall2(.mmap, len, prot);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Release a range previously handed out by `mmap`.
|
/// Release a range previously handed out by `mmap`.
|
||||||
pub fn munmap(base: usize, len: usize) usize {
|
pub fn munmap(base: usize, len: usize) usize {
|
||||||
return sc.syscall2(.munmap, base, len);
|
return sc.systemCall2(.munmap, base, len);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||||
+38
-38
@@ -4,13 +4,13 @@
|
|||||||
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const proto = @import("vfs_proto.zig");
|
const protocol = @import("vfs-protocol.zig");
|
||||||
const ipc = @import("ipc.zig");
|
const ipc = @import("ipc.zig");
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
|
|
||||||
pub const O_CREAT = proto.O_CREAT;
|
pub const O_CREAT = protocol.O_CREAT;
|
||||||
pub const SEEK_SET: u32 = 0;
|
pub const SEEK_SET: u32 = 0;
|
||||||
pub const SEEK_CUR: u32 = 1;
|
pub const SEEK_CURRENT: u32 = 1;
|
||||||
pub const SEEK_END: u32 = 2;
|
pub const SEEK_END: u32 = 2;
|
||||||
|
|
||||||
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
||||||
@@ -24,9 +24,9 @@ fn vfs() ?usize {
|
|||||||
return vfs_handle;
|
return vfs_handle;
|
||||||
}
|
}
|
||||||
|
|
||||||
const max_fds = 32;
|
const maximum_fds = 32;
|
||||||
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
||||||
var fds = [_]Fd{.{}} ** max_fds;
|
var fds = [_]Fd{.{}} ** maximum_fds;
|
||||||
|
|
||||||
fn allocFd() ?usize {
|
fn allocFd() ?usize {
|
||||||
for (&fds, 0..) |*f, i| {
|
for (&fds, 0..) |*f, i| {
|
||||||
@@ -38,30 +38,30 @@ fn allocFd() ?usize {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
const Result = struct { reply: proto.Reply, payload: []u8 };
|
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||||
|
|
||||||
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||||
/// [Reply header][recv payload]. The recv payload is written into `out`.
|
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
||||||
fn transact(req: proto.Request, send: []const u8, out: []u8) ?Result {
|
fn transact(req: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||||
const h = vfs() orelse return null;
|
const h = vfs() orelse return null;
|
||||||
var msg: [proto.msg_max]u8 = undefined;
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
@memcpy(msg[0..proto.req_size], std.mem.asBytes(&req));
|
@memcpy(message[0..protocol.req_size], std.mem.asBytes(&req));
|
||||||
const slen = @min(send.len, proto.max_payload);
|
const slen = @min(send.len, protocol.maximum_payload);
|
||||||
@memcpy(msg[proto.req_size..][0..slen], send[0..slen]);
|
@memcpy(message[protocol.req_size..][0..slen], send[0..slen]);
|
||||||
|
|
||||||
var rbuf: [proto.msg_max]u8 = undefined;
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
const n = ipc.call(h, msg[0 .. proto.req_size + slen], &rbuf) catch return null;
|
const n = ipc.call(h, message[0 .. protocol.req_size + slen], &rbuf) catch return null;
|
||||||
if (n < proto.reply_size) return null;
|
if (n < protocol.reply_size) return null;
|
||||||
const reply = std.mem.bytesToValue(proto.Reply, rbuf[0..proto.reply_size]);
|
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||||
const rpl = @min(n - proto.reply_size, out.len);
|
const rpl = @min(n - protocol.reply_size, out.len);
|
||||||
@memcpy(out[0..rpl], rbuf[proto.reply_size..][0..rpl]);
|
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
||||||
pub fn open(path: []const u8, flags: u32) i32 {
|
pub fn open(path: []const u8, flags: u32) i32 {
|
||||||
const fd = allocFd() orelse return -1;
|
const fd = allocFd() orelse return -1;
|
||||||
const req = proto.Request{ .op = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
const req = protocol.Request{ .op = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
||||||
const r = transact(req, path, &.{}) orelse {
|
const r = transact(req, path, &.{}) orelse {
|
||||||
fds[fd].used = false;
|
fds[fd].used = false;
|
||||||
return -1;
|
return -1;
|
||||||
@@ -75,17 +75,17 @@ pub fn open(path: []const u8, flags: u32) i32 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn fdPtr(fd: i32) ?*Fd {
|
fn fdPtr(fd: i32) ?*Fd {
|
||||||
if (fd < 0 or fd >= max_fds) return null;
|
if (fd < 0 or fd >= maximum_fds) return null;
|
||||||
const f = &fds[@intCast(fd)];
|
const f = &fds[@intCast(fd)];
|
||||||
return if (f.used) f else null;
|
return if (f.used) f else null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read up to `buf.len` bytes at the current offset; returns the count or -1.
|
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
||||||
pub fn read(fd: i32, buf: []u8) isize {
|
pub fn read(fd: i32, buffer: []u8) isize {
|
||||||
const f = fdPtr(fd) orelse return -1;
|
const f = fdPtr(fd) orelse return -1;
|
||||||
const want: u32 = @intCast(@min(buf.len, proto.max_payload));
|
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||||
const req = proto.Request{ .op = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
const req = protocol.Request{ .op = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||||
const r = transact(req, &.{}, buf) orelse return -1;
|
const r = transact(req, &.{}, buffer) orelse return -1;
|
||||||
if (r.reply.status != 0) return -1;
|
if (r.reply.status != 0) return -1;
|
||||||
f.offset += r.reply.len;
|
f.offset += r.reply.len;
|
||||||
return @intCast(r.reply.len);
|
return @intCast(r.reply.len);
|
||||||
@@ -94,8 +94,8 @@ pub fn read(fd: i32, buf: []u8) isize {
|
|||||||
/// Write `data` at the current offset; returns the count or -1.
|
/// Write `data` at the current offset; returns the count or -1.
|
||||||
pub fn write(fd: i32, data: []const u8) isize {
|
pub fn write(fd: i32, data: []const u8) isize {
|
||||||
const f = fdPtr(fd) orelse return -1;
|
const f = fdPtr(fd) orelse return -1;
|
||||||
const want: u32 = @intCast(@min(data.len, proto.max_payload));
|
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||||
const req = proto.Request{ .op = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
const req = protocol.Request{ .op = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||||
const r = transact(req, data[0..want], &.{}) orelse return -1;
|
const r = transact(req, data[0..want], &.{}) orelse return -1;
|
||||||
if (r.reply.status != 0) return -1;
|
if (r.reply.status != 0) return -1;
|
||||||
f.offset += r.reply.len;
|
f.offset += r.reply.len;
|
||||||
@@ -108,13 +108,13 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
|||||||
const f = fdPtr(fd) orelse return -1;
|
const f = fdPtr(fd) orelse return -1;
|
||||||
const base: i64 = switch (whence) {
|
const base: i64 = switch (whence) {
|
||||||
SEEK_SET => 0,
|
SEEK_SET => 0,
|
||||||
SEEK_CUR => @intCast(f.offset),
|
SEEK_CURRENT => @intCast(f.offset),
|
||||||
SEEK_END => blk: {
|
SEEK_END => blk: {
|
||||||
const req = proto.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
const req = protocol.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
var sbuf: [@sizeOf(proto.Stat)]u8 = undefined;
|
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
|
||||||
const r = transact(req, &.{}, &sbuf) orelse return -1;
|
const r = transact(req, &.{}, &sbuf) orelse return -1;
|
||||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(proto.Stat)) return -1;
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
|
||||||
const st = std.mem.bytesToValue(proto.Stat, sbuf[0..@sizeOf(proto.Stat)]);
|
const st = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
|
||||||
break :blk @intCast(st.size);
|
break :blk @intCast(st.size);
|
||||||
},
|
},
|
||||||
else => return -1,
|
else => return -1,
|
||||||
@@ -126,24 +126,24 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Stat `path`. Returns 0 or -1.
|
/// Stat `path`. Returns 0 or -1.
|
||||||
pub fn stat(path: []const u8, out: *proto.Stat) i32 {
|
pub fn stat(path: []const u8, out: *protocol.Stat) i32 {
|
||||||
// Open, stat by node, close — simple and enough for now.
|
// Open, stat by node, close — simple and enough for now.
|
||||||
const fd = open(path, 0);
|
const fd = open(path, 0);
|
||||||
if (fd < 0) return -1;
|
if (fd < 0) return -1;
|
||||||
defer close(fd);
|
defer close(fd);
|
||||||
const f = fdPtr(fd).?;
|
const f = fdPtr(fd).?;
|
||||||
const req = proto.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
const req = protocol.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
var sbuf: [@sizeOf(proto.Stat)]u8 = undefined;
|
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
|
||||||
const r = transact(req, &.{}, &sbuf) orelse return -1;
|
const r = transact(req, &.{}, &sbuf) orelse return -1;
|
||||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(proto.Stat)) return -1;
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
|
||||||
out.* = std.mem.bytesToValue(proto.Stat, sbuf[0..@sizeOf(proto.Stat)]);
|
out.* = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Close an fd (best effort — tells the VFS to release the open file).
|
/// Close an fd (best effort — tells the VFS to release the open file).
|
||||||
pub fn close(fd: i32) void {
|
pub fn close(fd: i32) void {
|
||||||
const f = fdPtr(fd) orelse return;
|
const f = fdPtr(fd) orelse return;
|
||||||
const req = proto.Request{ .op = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
const req = protocol.Request{ .op = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
_ = transact(req, &.{}, &.{});
|
_ = transact(req, &.{}, &.{});
|
||||||
f.used = false;
|
f.used = false;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
//! The VFS wire protocol — the message format spoken between a client (via the
|
//! The VFS wire protocol — the message format spoken between a client (via the
|
||||||
//! `rt` file API) and the user-space VFS server over IPC. A request is a fixed
|
//! `runtime` file API) and the user-space VFS server over IPC. A request is a fixed
|
||||||
//! `Request` header followed by an inline payload (a path, or write bytes); a
|
//! `Request` header followed by an inline payload (a path, or write bytes); a
|
||||||
//! reply is a fixed `Reply` header followed by an inline payload (read bytes, or
|
//! reply is a fixed `Reply` header followed by an inline payload (read bytes, or
|
||||||
//! a Stat). Everything fits in one IPC message (<= ipc MSG_MAX = 256 bytes).
|
//! a Stat). Everything fits in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
|
||||||
//!
|
//!
|
||||||
//! This is user-space only — the kernel knows nothing of files or paths; it only
|
//! This is user-space only — the kernel knows nothing of files or paths; it only
|
||||||
//! moves the bytes. Shared by lib/unistd.zig (client) and sbin/vfs.zig (server).
|
//! moves the bytes. Shared by lib/unistd.zig (client) and sbin/vfs.zig (server).
|
||||||
@@ -43,11 +43,11 @@ pub const Stat = extern struct {
|
|||||||
_pad: u32 = 0,
|
_pad: u32 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub const msg_max: usize = 256;
|
pub const message_maximum: usize = 256;
|
||||||
pub const req_size: usize = @sizeOf(Request);
|
pub const req_size: usize = @sizeOf(Request);
|
||||||
pub const reply_size: usize = @sizeOf(Reply);
|
pub const reply_size: usize = @sizeOf(Reply);
|
||||||
/// Largest inline payload that still fits one IPC message alongside a header.
|
/// Largest inline payload that still fits one IPC message alongside a header.
|
||||||
pub const max_payload: usize = msg_max - req_size;
|
pub const maximum_payload: usize = message_maximum - req_size;
|
||||||
|
|
||||||
/// Open flags.
|
/// Open flags.
|
||||||
pub const O_CREAT: u32 = 1;
|
pub const O_CREAT: u32 = 1;
|
||||||
+210
@@ -0,0 +1,210 @@
|
|||||||
|
//! /sbin/busd — a user-space **bus driver**, and the smallest honest example of one.
|
||||||
|
//!
|
||||||
|
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
||||||
|
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
||||||
|
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
||||||
|
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
||||||
|
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
||||||
|
//!
|
||||||
|
//! It's a toy bus, but nothing about the mechanism is: `busd` reads how many children
|
||||||
|
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
||||||
|
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
||||||
|
//! every one of those resources is contained in what `busd` was granted. A comparator
|
||||||
|
//! driver then claims a child and maps only *its* registers — not the whole block.
|
||||||
|
//!
|
||||||
|
//! It also proves the negative: registering a child whose window escapes the parent's
|
||||||
|
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
||||||
|
//! arbitrary physical memory.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const device = runtime.device;
|
||||||
|
|
||||||
|
const register_general_cap = 0x000;
|
||||||
|
|
||||||
|
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
||||||
|
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
||||||
|
return .{
|
||||||
|
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||||
|
.start = hpet_base + 0x100 + 0x20 * n,
|
||||||
|
.len = 0x20,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
const n = @min(total, buffer.len);
|
||||||
|
for (buffer[0..n]) |d| {
|
||||||
|
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||||
|
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
||||||
|
for (0..d.resource_count) |j| {
|
||||||
|
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The parent's MMIO resource, and its IRQ if it has one.
|
||||||
|
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
||||||
|
var mmio: device.ResourceDescriptor = undefined;
|
||||||
|
var irq: ?device.ResourceDescriptor = null;
|
||||||
|
for (0..d.resource_count) |j| {
|
||||||
|
const r = d.resources[j];
|
||||||
|
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
||||||
|
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
||||||
|
}
|
||||||
|
return .{ .mmio = mmio, .irq = irq };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
||||||
|
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
|
if (d.parent == parent_id) return d.id;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
|
_ = runtime.system.write("busd: out of memory\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
const parent = findHpet(buffer) orelse {
|
||||||
|
_ = runtime.system.write("busd: no HPET\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const resource = resourcesOf(parent);
|
||||||
|
|
||||||
|
// Claim the bus. Everything below is subdivision of what this claim granted.
|
||||||
|
//
|
||||||
|
// Claims are exclusive, and at a normal boot the kernel spawns every initrd
|
||||||
|
// binary — so hpetd may own the HPET already. That's not an error, it's the
|
||||||
|
// capability model working: exit quietly and leave the device to its owner. The
|
||||||
|
// `bus` test spawns busd alone, so there it wins the claim.
|
||||||
|
if (!device.claim(parent.id)) {
|
||||||
|
_ = runtime.system.write("busd: HPET already claimed by another driver, nothing to do\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Enumerate the bus: ask the hardware how many children it has.
|
||||||
|
const base = device.mmioMap(parent.id, 0) orelse {
|
||||||
|
_ = runtime.system.write("busd: mmio_map failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
||||||
|
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
||||||
|
|
||||||
|
// Publish one child per comparator, each owning only its own window.
|
||||||
|
var published: u64 = 0;
|
||||||
|
var n: u64 = 0;
|
||||||
|
while (n < n_children) : (n += 1) {
|
||||||
|
var child = std.mem.zeroes(device.DeviceDescriptor);
|
||||||
|
child.class = @intFromEnum(device.DeviceClass.timer);
|
||||||
|
child.hid_len = 6;
|
||||||
|
child.hid[0..6].* = "hpet-t".*;
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = timerWindow(resource.mmio.start, n);
|
||||||
|
// Comparators share the block's interrupt line; only one child can bind it,
|
||||||
|
// but all of them may legitimately name it.
|
||||||
|
if (resource.irq) |i| {
|
||||||
|
child.resources[child.resource_count] = i;
|
||||||
|
child.resource_count += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (device.register(parent.id, &child) == null) {
|
||||||
|
_ = runtime.system.write("busd: register failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
published += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The negative case. A window one byte past the end of the parent's must be
|
||||||
|
// refused — otherwise device_register would be "map any physical page you like".
|
||||||
|
// Confirm the table did not grow, not merely that the call returned null: null
|
||||||
|
// also means NoSpace/BadParent, so a size check is what actually proves the
|
||||||
|
// *containment* rule fired.
|
||||||
|
const before = device.enumerate(buffer);
|
||||||
|
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
||||||
|
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
||||||
|
rogue.resource_count = 1;
|
||||||
|
rogue.resources[0] = .{
|
||||||
|
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||||
|
.start = resource.mmio.start + resource.mmio.len,
|
||||||
|
.len = 0x1000,
|
||||||
|
};
|
||||||
|
if (device.register(parent.id, &rogue) != null) {
|
||||||
|
_ = runtime.system.write("busd: FAIL out-of-window child was accepted\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (device.enumerate(buffer) != before) {
|
||||||
|
_ = runtime.system.write("busd: FAIL rogue child leaked into the table\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// And confirm the children came back with the right parent and a *narrower*
|
||||||
|
// window than the bus — read from the table, not from our own memory.
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
var seen: u64 = 0;
|
||||||
|
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
|
if (d.parent != parent.id) continue;
|
||||||
|
const w = d.resources[0];
|
||||||
|
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
||||||
|
_ = runtime.system.write("busd: FAIL child window is not inside the bus\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
seen += 1;
|
||||||
|
}
|
||||||
|
if (seen != published) {
|
||||||
|
_ = runtime.system.write("busd: FAIL child count mismatch\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
||||||
|
// a different process; here busd plays both parts, which exercises the same path.
|
||||||
|
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
||||||
|
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
||||||
|
//
|
||||||
|
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
||||||
|
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
||||||
|
// The *resource* is narrow even though the page isn't.)
|
||||||
|
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
||||||
|
_ = runtime.system.write("busd: FAIL no child to claim\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (!device.claim(child_id)) {
|
||||||
|
_ = runtime.system.write("busd: FAIL could not claim own child\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const child_base = device.mmioMap(child_id, 0) orelse {
|
||||||
|
_ = runtime.system.write("busd: FAIL child mmio_map refused\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
||||||
|
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
||||||
|
if (via_child.* != via_bus.*) {
|
||||||
|
_ = runtime.system.write("busd: FAIL child window does not alias the bus register\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
||||||
|
// kernel. Grab a page, free it, and register through the stale address: if the
|
||||||
|
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
||||||
|
// would triple-fault QEMU and the test would time out instead of printing ok.
|
||||||
|
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
||||||
|
if (!runtime.system.mmapFailed(scratch)) {
|
||||||
|
_ = runtime.system.munmap(scratch, 0x1000);
|
||||||
|
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
||||||
|
if (device.register(parent.id, descriptor) != null) {
|
||||||
|
_ = runtime.system.write("busd: FAIL register accepted an unmapped descriptor\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
_ = runtime.system.write("busd: ok\n");
|
||||||
|
while (true) runtime.system.sleep(1000);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
+172
-60
@@ -1,75 +1,187 @@
|
|||||||
//! /sbin/hpetd — a user-space HPET driver, the first real device driver. It
|
//! /sbin/hpetd — a user-space HPET driver. It proves the whole driver model end to
|
||||||
//! proves IO passthrough end to end: enumerate the device table, find the HPET
|
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
||||||
//! (a timer with an MMIO window), claim it, map its registers directly into this
|
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
||||||
//! ring-3 address space (strong-uncacheable), then drive the hardware — enable
|
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
||||||
//! the main counter and read it. If the counter advances, a user process is
|
|
||||||
//! touching real hardware through a kernel-granted MMIO mapping.
|
|
||||||
//!
|
//!
|
||||||
//! Register offsets (HPET spec): general config = 0x10 (bit 0 = ENABLE),
|
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
||||||
//! main counter = 0xF0.
|
//! scheduler queue; the core runs other work or idles. That is the point of the
|
||||||
|
//! exercise — a driver is a process that sleeps until its device has something to
|
||||||
|
//! say (see docs/drivers.md).
|
||||||
|
//!
|
||||||
|
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
||||||
|
//! simpler, but level is the discipline every real device line needs, and it forces
|
||||||
|
//! the full cycle to be correct:
|
||||||
|
//!
|
||||||
|
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
||||||
|
//! hpetd wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||||
|
//! hpetd irq_ack -> kernel unmasks the GSI
|
||||||
|
//!
|
||||||
|
//! Clear the status bit *before* acking, or the line is still asserted when the
|
||||||
|
//! kernel unmasks and the I/O APIC redelivers forever.
|
||||||
|
//!
|
||||||
|
//! Register map (HPET spec 1.0a):
|
||||||
|
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
||||||
|
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
||||||
|
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
||||||
|
//! 0x0F0 MAIN_COUNTER
|
||||||
|
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
||||||
|
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
||||||
|
//! 0x108 TIMER0_COMPARATOR
|
||||||
|
|
||||||
const rt = @import("rt");
|
const runtime = @import("runtime");
|
||||||
const dev = rt.dev;
|
const device = runtime.device;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
|
||||||
|
const register_general_cap = 0x000;
|
||||||
|
const register_general_configuration = 0x010;
|
||||||
|
const register_int_status = 0x020;
|
||||||
|
const register_main_counter = 0x0F0;
|
||||||
|
const register_timer0_configuration = 0x100;
|
||||||
|
const register_timer0_comparator = 0x108;
|
||||||
|
|
||||||
|
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
||||||
|
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
||||||
|
const tn_int_type_level: u64 = 1 << 1;
|
||||||
|
const tn_int_enb: u64 = 1 << 2;
|
||||||
|
const tn_type_periodic: u64 = 1 << 3;
|
||||||
|
const tn_route_shift = 9;
|
||||||
|
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
||||||
|
|
||||||
|
/// Interrupts to observe before declaring victory.
|
||||||
|
const target_ticks = 5;
|
||||||
|
|
||||||
|
fn register(base: usize, off: usize) *volatile u64 {
|
||||||
|
return @ptrFromInt(base + off);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
||||||
|
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
||||||
|
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
||||||
|
|
||||||
|
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
const n = @min(total, buffer.len);
|
||||||
|
for (buffer[0..n]) |d| {
|
||||||
|
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||||
|
// Skip comparator children a bus driver may have published below the block
|
||||||
|
// (see sbin/busd.zig) — we want the register block itself.
|
||||||
|
if (d.parent != device.no_parent) continue;
|
||||||
|
var mmio: ?u64 = null;
|
||||||
|
var irq: ?u64 = null;
|
||||||
|
for (0..d.resource_count) |j| {
|
||||||
|
switch (d.resources[j].kind) {
|
||||||
|
@intFromEnum(device.ResourceKind.memory) => mmio = mmio orelse j,
|
||||||
|
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (mmio) |m| if (irq) |i| {
|
||||||
|
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
||||||
|
};
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||||
const buf = rt.allocator().alloc(dev.DeviceDesc, 32) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
||||||
_ = rt.sys.write("hpetd: out of memory\n");
|
_ = runtime.system.write("hpetd: out of memory\n");
|
||||||
return;
|
|
||||||
};
|
|
||||||
const total = dev.enumerate(buf);
|
|
||||||
const n = @min(total, buf.len);
|
|
||||||
|
|
||||||
// Find a timer-class device with an MMIO resource (the HPET).
|
|
||||||
var dev_id: u64 = 0;
|
|
||||||
var res_idx: u64 = 0;
|
|
||||||
var found = false;
|
|
||||||
var i: usize = 0;
|
|
||||||
outer: while (i < n) : (i += 1) {
|
|
||||||
const d = buf[i];
|
|
||||||
if (d.class != @intFromEnum(dev.DeviceClass.timer)) continue;
|
|
||||||
var j: usize = 0;
|
|
||||||
while (j < d.resource_count) : (j += 1) {
|
|
||||||
if (d.resources[j].kind == @intFromEnum(dev.ResourceKind.memory)) {
|
|
||||||
dev_id = d.id;
|
|
||||||
res_idx = j;
|
|
||||||
found = true;
|
|
||||||
break :outer;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!found) {
|
|
||||||
_ = rt.sys.write("hpetd: no HPET found\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!dev.claim(dev_id)) {
|
|
||||||
_ = rt.sys.write("hpetd: claim failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const base = dev.mmioMap(dev_id, res_idx) orelse {
|
|
||||||
_ = rt.sys.write("hpetd: mmio_map failed\n");
|
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Drive the hardware: enable the counter (an MMIO write), then read it twice.
|
const hpet = findHpet(buffer) orelse {
|
||||||
const config: *volatile u64 = @ptrFromInt(base + 0x10);
|
_ = runtime.system.write("hpetd: no HPET with an IRQ\n");
|
||||||
config.* |= 1; // ENABLE
|
return;
|
||||||
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
};
|
||||||
const a = counter.*;
|
|
||||||
rt.sys.sleep(50);
|
|
||||||
const b = counter.*;
|
|
||||||
|
|
||||||
if (b > a) {
|
if (!device.claim(hpet.device_id)) {
|
||||||
while (true) {
|
_ = runtime.system.write("hpetd: claim failed\n");
|
||||||
_ = rt.sys.write("hpetd: ok\n");
|
return;
|
||||||
rt.sys.sleep(1000);
|
}
|
||||||
|
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
||||||
|
_ = runtime.system.write("hpetd: mmio_map failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
||||||
|
// to raise exactly this line — the kernel will only bind the one it recorded.
|
||||||
|
const gsi = hpet.gsi;
|
||||||
|
|
||||||
|
const endpoint = ipc.createEndpoint() orelse {
|
||||||
|
_ = runtime.system.write("hpetd: create_endpoint failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- program the hardware ------------------------------------------------
|
||||||
|
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
||||||
|
const femtos_per_tick = register(base, register_general_cap).* >> 32;
|
||||||
|
if (femtos_per_tick == 0) {
|
||||||
|
_ = runtime.system.write("hpetd: bad HPET period\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
||||||
|
|
||||||
|
// Stop the counter and take the legacy route off while we reconfigure.
|
||||||
|
register(base, register_general_configuration).* &= ~(configuration_enable | configuration_leg_rt);
|
||||||
|
|
||||||
|
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
||||||
|
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
||||||
|
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
||||||
|
// timer driver does anyway.
|
||||||
|
var t0 = register(base, register_timer0_configuration).*;
|
||||||
|
t0 &= ~(tn_route_mask | tn_type_periodic);
|
||||||
|
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
||||||
|
register(base, register_timer0_configuration).* = t0;
|
||||||
|
|
||||||
|
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
||||||
|
register(base, register_int_status).* = 1;
|
||||||
|
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
|
||||||
|
register(base, register_general_configuration).* |= configuration_enable;
|
||||||
|
|
||||||
|
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
||||||
|
_ = runtime.system.write("hpetd: irq_bind failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = runtime.system.write("hpetd: bound, sleeping until the hardware speaks\n");
|
||||||
|
|
||||||
|
// --- the driver loop -----------------------------------------------------
|
||||||
|
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
||||||
|
// runs only because an interrupt fired.
|
||||||
|
var receive: [64]u8 = undefined;
|
||||||
|
var count: usize = 0;
|
||||||
|
while (count < target_ticks) {
|
||||||
|
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
||||||
|
// next line runs only because the HPET raised its line.
|
||||||
|
const r = ipc.replyWait(endpoint, &.{}, &receive);
|
||||||
|
if (!r.isNotification()) continue; // a client request, not our IRQ
|
||||||
|
|
||||||
|
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
||||||
|
// line is still asserted and unmasking would refire immediately.
|
||||||
|
register(base, register_int_status).* = 1;
|
||||||
|
count += 1;
|
||||||
|
|
||||||
|
if (count < target_ticks) {
|
||||||
|
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
|
||||||
|
} else {
|
||||||
|
// Last one: stop the source rather than re-arming, so the line is left
|
||||||
|
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
||||||
|
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
||||||
|
// the line again a moment later.
|
||||||
|
register(base, register_timer0_configuration).* &= ~tn_int_enb;
|
||||||
|
}
|
||||||
|
|
||||||
|
_ = runtime.system.write("hpetd: irq\n");
|
||||||
|
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
||||||
|
_ = runtime.system.write("hpetd: irq_ack failed\n");
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
_ = rt.sys.write("hpetd: counter stuck\n");
|
|
||||||
|
_ = runtime.system.write("hpetd: ok\n");
|
||||||
|
while (true) runtime.system.sleep(1000);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = rt.panic;
|
pub const panic = runtime.panic;
|
||||||
comptime {
|
comptime {
|
||||||
_ = &rt.start._start;
|
_ = &runtime.start._start;
|
||||||
}
|
}
|
||||||
|
|||||||
+12
-12
@@ -2,7 +2,7 @@
|
|||||||
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
|
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
|
||||||
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
|
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
|
||||||
//! kernel (src/kernel/process.zig). It links against the shared user runtime
|
//! kernel (src/kernel/process.zig). It links against the shared user runtime
|
||||||
//! library `rt` and talks to the kernel only through `rt`'s syscall wrappers.
|
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
|
||||||
//!
|
//!
|
||||||
//! Today it proves the C-convention heap works, then settles into a heartbeat:
|
//! Today it proves the C-convention heap works, then settles into a heartbeat:
|
||||||
//! it prints a line and sleeps, forever — enough to show the system reaches user
|
//! it prints a line and sleeps, forever — enough to show the system reaches user
|
||||||
@@ -10,7 +10,7 @@
|
|||||||
//! idle loop. It grows into the real init (service supervision) once there are
|
//! idle loop. It grows into the real init (service supervision) once there are
|
||||||
//! other user programs to supervise.
|
//! other user programs to supervise.
|
||||||
|
|
||||||
const rt = @import("rt");
|
const runtime = @import("runtime");
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||||
@@ -19,21 +19,21 @@ pub fn main() void {
|
|||||||
// free it. A fault here would kill init before it heartbeats — so the init
|
// free it. A fault here would kill init before it heartbeats — so the init
|
||||||
// test doubles as the heap regression test. (C code links the same heap via
|
// test doubles as the heap regression test. (C code links the same heap via
|
||||||
// the extern malloc/free symbols; Zig code uses this allocator.)
|
// the extern malloc/free symbols; Zig code uses this allocator.)
|
||||||
const gpa = rt.allocator();
|
const gpa = runtime.allocator();
|
||||||
if (gpa.alloc(u8, 64)) |buf| {
|
if (gpa.alloc(u8, 64)) |buffer| {
|
||||||
const msg = "init: heap ok\n";
|
const message = "init: heap ok\n";
|
||||||
@memcpy(buf[0..msg.len], msg);
|
@memcpy(buffer[0..message.len], message);
|
||||||
_ = rt.sys.write(buf[0..msg.len]);
|
_ = runtime.system.write(buffer[0..message.len]);
|
||||||
gpa.free(buf);
|
gpa.free(buffer);
|
||||||
} else |_| {}
|
} else |_| {}
|
||||||
|
|
||||||
while (true) {
|
while (true) {
|
||||||
_ = rt.sys.write("init: heartbeat\n");
|
_ = runtime.system.write("init: heartbeat\n");
|
||||||
rt.sys.sleep(1000);
|
runtime.system.sleep(1000);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = rt.panic;
|
pub const panic = runtime.panic;
|
||||||
comptime {
|
comptime {
|
||||||
_ = &rt.start._start; // pull the runtime entry shim into the image
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,13 +1,13 @@
|
|||||||
//! /sbin/vfstest — a client that proves the VFS round trip end to end: open a
|
//! /sbin/vfstest — a client that proves the VFS round trip end to end: open a
|
||||||
//! file through the `rt` file API, write to it, seek back, read it, and compare.
|
//! file through the `runtime` file API, write to it, seek back, read it, and compare.
|
||||||
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
|
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
|
||||||
//! on failure it reports what went wrong. Shipped in the initrd alongside vfs.
|
//! on failure it reports what went wrong. Shipped in the initrd alongside vfs.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const rt = @import("rt");
|
const runtime = @import("runtime");
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
const u = rt.unistd;
|
const u = runtime.unistd;
|
||||||
const payload = "hello-vfs";
|
const payload = "hello-vfs";
|
||||||
|
|
||||||
// The VFS server may not have registered yet — retry open until it's up.
|
// The VFS server may not have registered yet — retry open until it's up.
|
||||||
@@ -15,33 +15,33 @@ pub fn main() void {
|
|||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
while (fd < 0 and tries < 200) : (tries += 1) {
|
while (fd < 0 and tries < 200) : (tries += 1) {
|
||||||
fd = u.open("greeting", u.O_CREAT);
|
fd = u.open("greeting", u.O_CREAT);
|
||||||
if (fd < 0) rt.sys.sleep(20);
|
if (fd < 0) runtime.system.sleep(20);
|
||||||
}
|
}
|
||||||
if (fd < 0) {
|
if (fd < 0) {
|
||||||
_ = rt.sys.write("vfstest: open failed\n");
|
_ = runtime.system.write("vfstest: open failed\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (u.write(fd, payload) != @as(isize, payload.len)) {
|
if (u.write(fd, payload) != @as(isize, payload.len)) {
|
||||||
_ = rt.sys.write("vfstest: write failed\n");
|
_ = runtime.system.write("vfstest: write failed\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
_ = u.lseek(fd, 0, u.SEEK_SET);
|
_ = u.lseek(fd, 0, u.SEEK_SET);
|
||||||
|
|
||||||
var buf: [32]u8 = undefined;
|
var buffer: [32]u8 = undefined;
|
||||||
const n = u.read(fd, &buf);
|
const n = u.read(fd, &buffer);
|
||||||
u.close(fd);
|
u.close(fd);
|
||||||
|
|
||||||
if (n == @as(isize, payload.len) and std.mem.eql(u8, buf[0..@intCast(n)], payload)) {
|
if (n == @as(isize, payload.len) and std.mem.eql(u8, buffer[0..@intCast(n)], payload)) {
|
||||||
while (true) {
|
while (true) {
|
||||||
_ = rt.sys.write("vfstest: ok\n");
|
_ = runtime.system.write("vfstest: ok\n");
|
||||||
rt.sys.sleep(1000);
|
runtime.system.sleep(1000);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
_ = rt.sys.write("vfstest: mismatch\n");
|
_ = runtime.system.write("vfstest: mismatch\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = rt.panic;
|
pub const panic = runtime.panic;
|
||||||
comptime {
|
comptime {
|
||||||
_ = &rt.start._start;
|
_ = &runtime.start._start;
|
||||||
}
|
}
|
||||||
+28
-28
@@ -1,16 +1,16 @@
|
|||||||
//! /sbin/vfs — the user-space VFS server. Shipped in the initrd, spawned as a
|
//! /sbin/vfs — the user-space VFS server. Shipped in the initrd, spawned as a
|
||||||
//! ring-3 process, and reached by every other process through IPC (the `rt`
|
//! ring-3 process, and reached by every other process through IPC (the `runtime`
|
||||||
//! file API marshals open/read/write/stat/close into calls to this server's
|
//! file API marshals open/read/write/stat/close into calls to this server's
|
||||||
//! endpoint, published under the well-known `vfs` service id).
|
//! endpoint, published under the well-known `vfs` service id).
|
||||||
//!
|
//!
|
||||||
//! For now the namespace is a small in-memory ramfs (opening a name creates it):
|
//! For now the namespace is a small in-memory ramfs (opening a name creates it):
|
||||||
//! enough to prove the whole path — client file API -> IPC -> server dispatch ->
|
//! enough to prove the whole path — client file API -> IPC -> server dispatch ->
|
||||||
//! reply. Device nodes backed by user-space drivers (/dev) layer on top in M10,
|
//! reply. Device nodes backed by user-space drivers (/device) layer on top in M10,
|
||||||
//! where `open` on a /dev name forwards to the owning driver's endpoint.
|
//! where `open` on a /device name forwards to the owning driver's endpoint.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const rt = @import("rt");
|
const runtime = @import("runtime");
|
||||||
const proto = rt.vfsproto;
|
const protocol = runtime.vfs_protocol;
|
||||||
|
|
||||||
const Node = struct {
|
const Node = struct {
|
||||||
used: bool = false,
|
used: bool = false,
|
||||||
@@ -54,11 +54,11 @@ fn openAt(id: u64) ?*OpenFile {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Serialise a reply header + payload into `out`; returns the total length.
|
/// Serialise a reply header + payload into `out`; returns the total length.
|
||||||
fn writeReply(out: []u8, reply: proto.Reply, payload: []const u8) usize {
|
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
|
||||||
@memcpy(out[0..proto.reply_size], std.mem.asBytes(&reply));
|
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
|
||||||
const n = @min(payload.len, out.len - proto.reply_size);
|
const n = @min(payload.len, out.len - protocol.reply_size);
|
||||||
@memcpy(out[proto.reply_size..][0..n], payload[0..n]);
|
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
|
||||||
return proto.reply_size + n;
|
return protocol.reply_size + n;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn fail(out: []u8) usize {
|
fn fail(out: []u8) usize {
|
||||||
@@ -66,10 +66,10 @@ fn fail(out: []u8) usize {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Handle one request; write the reply into `out`, return its length.
|
/// Handle one request; write the reply into `out`, return its length.
|
||||||
fn handle(msg: []const u8, out: []u8) usize {
|
fn handle(message: []const u8, out: []u8) usize {
|
||||||
if (msg.len < proto.req_size) return fail(out);
|
if (message.len < protocol.req_size) return fail(out);
|
||||||
const req = std.mem.bytesToValue(proto.Request, msg[0..proto.req_size]);
|
const req = std.mem.bytesToValue(protocol.Request, message[0..protocol.req_size]);
|
||||||
const payload = msg[proto.req_size..];
|
const payload = message[protocol.req_size..];
|
||||||
|
|
||||||
switch (req.op) {
|
switch (req.op) {
|
||||||
.open => {
|
.open => {
|
||||||
@@ -88,7 +88,7 @@ fn handle(msg: []const u8, out: []u8) usize {
|
|||||||
const nd = &nodes[of.node];
|
const nd = &nodes[of.node];
|
||||||
const off: usize = @intCast(req.offset);
|
const off: usize = @intCast(req.offset);
|
||||||
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
|
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
|
||||||
const n = @min(@min(nd.size - off, req.len), proto.max_payload);
|
const n = @min(@min(nd.size - off, req.len), protocol.maximum_payload);
|
||||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
|
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
|
||||||
},
|
},
|
||||||
.write => {
|
.write => {
|
||||||
@@ -103,8 +103,8 @@ fn handle(msg: []const u8, out: []u8) usize {
|
|||||||
},
|
},
|
||||||
.stat => {
|
.stat => {
|
||||||
const of = openAt(req.node) orelse return fail(out);
|
const of = openAt(req.node) orelse return fail(out);
|
||||||
const st = proto.Stat{ .size = nodes[of.node].size, .kind = 0 };
|
const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 };
|
||||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(proto.Stat) }, std.mem.asBytes(&st));
|
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st));
|
||||||
},
|
},
|
||||||
.close => {
|
.close => {
|
||||||
if (req.node < opens.len) opens[@intCast(req.node)].used = false;
|
if (req.node < opens.len) opens[@intCast(req.node)].used = false;
|
||||||
@@ -114,27 +114,27 @@ fn handle(msg: []const u8, out: []u8) usize {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
const ep = rt.ipc.createEndpoint() orelse {
|
const endpoint = runtime.ipc.createEndpoint() orelse {
|
||||||
_ = rt.sys.write("vfs: no endpoint\n");
|
_ = runtime.system.write("vfs: no endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
if (!rt.ipc.register(.vfs, ep)) {
|
if (!runtime.ipc.register(.vfs, endpoint)) {
|
||||||
_ = rt.sys.write("vfs: register failed\n");
|
_ = runtime.system.write("vfs: register failed\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
_ = rt.sys.write("vfs: ready\n");
|
_ = runtime.system.write("vfs: ready\n");
|
||||||
|
|
||||||
var reply_buf: [proto.msg_max]u8 = undefined;
|
var reply_buffer: [protocol.message_maximum]u8 = undefined;
|
||||||
var reply_len: usize = 0;
|
var reply_len: usize = 0;
|
||||||
var recv: [proto.msg_max]u8 = undefined;
|
var receive: [protocol.message_maximum]u8 = undefined;
|
||||||
while (true) {
|
while (true) {
|
||||||
const got = rt.ipc.replyWait(ep, reply_buf[0..reply_len], &recv);
|
const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive);
|
||||||
// Ignore notifications (none expected here); handle a request.
|
// Ignore notifications (none expected here); handle a request.
|
||||||
reply_len = handle(recv[0..got.len], &reply_buf);
|
reply_len = handle(receive[0..got.len], &reply_buffer);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub const panic = rt.panic;
|
pub const panic = runtime.panic;
|
||||||
comptime {
|
comptime {
|
||||||
_ = &rt.start._start;
|
_ = &runtime.start._start;
|
||||||
}
|
}
|
||||||
|
|||||||
+73
-73
@@ -2,7 +2,7 @@ const std = @import("std");
|
|||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const BootInfo = danos.BootInfo;
|
const BootInformation = danos.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||||
@@ -40,7 +40,7 @@ fn boot() !noreturn {
|
|||||||
|
|
||||||
// Everything the kernel needs must be gathered *before* we exit boot
|
// Everything the kernel needs must be gathered *before* we exit boot
|
||||||
// services, since afterwards none of these calls are usable.
|
// services, since afterwards none of these calls are usable.
|
||||||
var boot_info: BootInfo = .{
|
var boot_information: BootInformation = .{
|
||||||
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||||
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||||
.framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{
|
.framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{
|
||||||
@@ -59,17 +59,17 @@ fn boot() !noreturn {
|
|||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const entry = try loadKernel(bs, &boot_info);
|
const entry = try loadKernel(bs, &boot_information);
|
||||||
|
|
||||||
// Best effort: a volume without sbin/init still boots (kernel-only).
|
// Best effort: a volume without sbin/init still boots (kernel-only).
|
||||||
loadInit(bs, &boot_info) catch |err| {
|
loadInit(bs, &boot_information) catch |err| {
|
||||||
log("danos: no sbin/init (");
|
log("danos: no sbin/init (");
|
||||||
logBytes(@errorName(err));
|
logBytes(@errorName(err));
|
||||||
log(") - booting without user space\r\n");
|
log(") - booting without user space\r\n");
|
||||||
};
|
};
|
||||||
|
|
||||||
// Best effort: the initrd (VFS server + drivers) is optional too.
|
// Best effort: the initrd (VFS server + drivers) is optional too.
|
||||||
loadInitrd(bs, &boot_info) catch |err| {
|
loadInitrd(bs, &boot_information) catch |err| {
|
||||||
log("danos: no initrd (");
|
log("danos: no initrd (");
|
||||||
logBytes(@errorName(err));
|
logBytes(@errorName(err));
|
||||||
log(")\r\n");
|
log(")\r\n");
|
||||||
@@ -80,17 +80,17 @@ fn boot() !noreturn {
|
|||||||
// now, while boot services (and the memory map) are still stable — nothing
|
// now, while boot services (and the memory map) are still stable — nothing
|
||||||
// is allocatable after ExitBootServices, and any allocation between fetching
|
// is allocatable after ExitBootServices, and any allocation between fetching
|
||||||
// the map and exiting would invalidate the map key.
|
// the map and exiting would invalidate the map key.
|
||||||
const cr3 = try buildBootstrapTables(bs, &boot_info);
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
log("danos: kernel loaded, exiting boot services\r\n");
|
log("danos: kernel loaded, exiting boot services\r\n");
|
||||||
boot_info.memory_map = try exitBootServices(bs);
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
// We load RDI explicitly (SysV first arg) rather than trusting this UEFI
|
// We load RDI explicitly (SystemV first arg) rather than trusting this UEFI
|
||||||
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
||||||
// higher-half) entry — the bootstrap tables map both the low loader code
|
// higher-half) entry — the bootstrap tables map both the low loader code
|
||||||
// executing this and the kernel's link address.
|
// executing this and the kernel's link address.
|
||||||
handoff(cr3, entry, &boot_info);
|
handoff(cr3, entry, &boot_information);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A display resolution in pixels.
|
/// A display resolution in pixels.
|
||||||
@@ -195,7 +195,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
|||||||
|
|
||||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||||
/// load its segments, and return the physical entry-point address.
|
/// load its segments, and return the physical entry-point address.
|
||||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize {
|
||||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
return error.NoLoadedImage;
|
return error.NoLoadedImage;
|
||||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
@@ -224,7 +224,7 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
|||||||
read_total += n;
|
read_total += n;
|
||||||
}
|
}
|
||||||
|
|
||||||
return loadElf(bs, image, boot_info);
|
return loadElf(bs, image, boot_information);
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- bootstrap page tables -------------------------------------------------
|
// --- bootstrap page tables -------------------------------------------------
|
||||||
@@ -241,7 +241,7 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
|||||||
const pte_present: u64 = 1 << 0;
|
const pte_present: u64 = 1 << 0;
|
||||||
const pte_write: u64 = 1 << 1;
|
const pte_write: u64 = 1 << 1;
|
||||||
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
||||||
const pte_addr: u64 = 0x000F_FFFF_FFFF_F000;
|
const pte_address: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
const gib: u64 = 1 << 30;
|
const gib: u64 = 1 << 30;
|
||||||
|
|
||||||
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
||||||
@@ -258,40 +258,40 @@ const TablePool = struct {
|
|||||||
return frame;
|
return frame;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn table(phys: u64) *[512]u64 {
|
fn table(physical: u64) *[512]u64 {
|
||||||
return @ptrFromInt(phys);
|
return @ptrFromInt(physical);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return the next-level table an entry points at, creating it if absent.
|
/// Return the next-level table an entry points at, creating it if absent.
|
||||||
fn descend(self: *TablePool, entry: *u64) !u64 {
|
fn descend(self: *TablePool, entry: *u64) !u64 {
|
||||||
if (entry.* & pte_present != 0) return entry.* & pte_addr;
|
if (entry.* & pte_present != 0) return entry.* & pte_address;
|
||||||
const frame = try self.alloc();
|
const frame = try self.alloc();
|
||||||
entry.* = frame | pte_present | pte_write;
|
entry.* = frame | pte_present | pte_write;
|
||||||
return frame;
|
return frame;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn map2M(self: *TablePool, pml4: u64, virt: u64, phys: u64) !void {
|
fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
const pml4e = &table(pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
const pdpt = try self.descend(pml4e);
|
const pdpt = try self.descend(pml4e);
|
||||||
const pdpte = &table(pdpt)[(virt >> 30) & 0x1FF];
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
const pd = try self.descend(pdpte);
|
const pd = try self.descend(pdpte);
|
||||||
table(pd)[(virt >> 21) & 0x1FF] = (phys & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn map4K(self: *TablePool, pml4: u64, virt: u64, phys: u64) !void {
|
fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
const pml4e = &table(pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
const pdpt = try self.descend(pml4e);
|
const pdpt = try self.descend(pml4e);
|
||||||
const pdpte = &table(pdpt)[(virt >> 30) & 0x1FF];
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
const pd = try self.descend(pdpte);
|
const pd = try self.descend(pdpte);
|
||||||
const pde = &table(pd)[(virt >> 21) & 0x1FF];
|
const pde = &table(pd)[(virtual >> 21) & 0x1FF];
|
||||||
const pt = try self.descend(pde);
|
const pt = try self.descend(pde);
|
||||||
table(pt)[(virt >> 12) & 0x1FF] = (phys & pte_addr) | pte_present | pte_write;
|
table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
||||||
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
||||||
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_info: *const BootInfo) !u64 {
|
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 {
|
||||||
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
||||||
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
||||||
const pool_pages = 64;
|
const pool_pages = 64;
|
||||||
@@ -303,44 +303,44 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_info: *const BootInf
|
|||||||
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
||||||
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
||||||
// framebuffer above 4 GiB would extend this — see the fb window below.
|
// framebuffer above 4 GiB would extend this — see the fb window below.
|
||||||
var addr: u64 = 0;
|
var address: u64 = 0;
|
||||||
while (addr < 4 * gib) : (addr += 2 << 20) {
|
while (address < 4 * gib) : (address += 2 << 20) {
|
||||||
try pool.map2M(pml4, addr, addr); // identity
|
try pool.map2M(pml4, address, address); // identity
|
||||||
try pool.map2M(pml4, danos.physToVirt(addr), addr); // physmap
|
try pool.map2M(pml4, danos.physicalToVirtual(address), address); // physmap
|
||||||
}
|
}
|
||||||
|
|
||||||
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||||
// pages (the kernel touches fb.base before it builds its own tables).
|
// pages (the kernel touches fb.base before it builds its own tables).
|
||||||
const fb = boot_info.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
||||||
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
||||||
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||||
while (p < fb_end) : (p += 2 << 20) {
|
while (p < fb_end) : (p += 2 << 20) {
|
||||||
try pool.map2M(pml4, p, p);
|
try pool.map2M(pml4, p, p);
|
||||||
try pool.map2M(pml4, danos.physToVirt(p), p);
|
try pool.map2M(pml4, danos.physicalToVirtual(p), p);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Higher-half kernel segments (virt != phys). While the kernel still links
|
// Higher-half kernel segments (virtual != physical). While the kernel still links
|
||||||
// low its segments sit in the identity range and need no separate mapping
|
// low its segments sit in the identity range and need no separate mapping
|
||||||
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||||
// only map segments that actually live in the higher half.
|
// only map segments that actually live in the higher half.
|
||||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||||
if (seg.virt < danos.kernel_virt_base) continue;
|
if (seg.virtual < danos.kernel_virt_base) continue;
|
||||||
var off: u64 = 0;
|
var off: u64 = 0;
|
||||||
while (off < seg.pages * page_size) : (off += page_size) {
|
while (off < seg.pages * page_size) : (off += page_size) {
|
||||||
try pool.map4K(pml4, seg.virt + off, seg.phys + off);
|
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return pml4;
|
return pml4;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_info` in RDI,
|
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI,
|
||||||
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
||||||
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
||||||
/// load; the jump target is mapped (identity while low, higher-half once high).
|
/// load; the jump target is mapped (identity while low, higher-half once high).
|
||||||
fn handoff(cr3: u64, entry: usize, boot_info: *const BootInfo) noreturn {
|
fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn {
|
||||||
asm volatile (
|
asm volatile (
|
||||||
\\cli
|
\\cli
|
||||||
\\movq %[cr3], %%cr3
|
\\movq %[cr3], %%cr3
|
||||||
@@ -348,7 +348,7 @@ fn handoff(cr3: u64, entry: usize, boot_info: *const BootInfo) noreturn {
|
|||||||
\\callq *%[entry]
|
\\callq *%[entry]
|
||||||
:
|
:
|
||||||
: [cr3] "r" (cr3),
|
: [cr3] "r" (cr3),
|
||||||
[bi] "r" (boot_info),
|
[bi] "r" (boot_information),
|
||||||
[entry] "r" (entry),
|
[entry] "r" (entry),
|
||||||
: .{ .memory = true });
|
: .{ .memory = true });
|
||||||
unreachable;
|
unreachable;
|
||||||
@@ -389,25 +389,25 @@ fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
|||||||
|
|
||||||
/// Ferry the init program (sbin/init) to the kernel. The kernel does the ELF
|
/// Ferry the init program (sbin/init) to the kernel. The kernel does the ELF
|
||||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||||
fn loadInit(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void {
|
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
const image = try loadFile(bs, init_file_name);
|
const image = try loadFile(bs, init_file_name);
|
||||||
boot_info.init_base = @intFromPtr(image.ptr);
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
boot_info.init_len = image.len;
|
boot_information.init_len = image.len;
|
||||||
log("danos: sbin/init loaded\r\n");
|
log("danos: sbin/init loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Ferry the initrd (the VFS server + drivers) to the kernel, same as init.
|
/// Ferry the initrd (the VFS server + drivers) to the kernel, same as init.
|
||||||
fn loadInitrd(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void {
|
fn loadInitrd(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
const image = try loadFile(bs, initrd_file_name);
|
const image = try loadFile(bs, initrd_file_name);
|
||||||
boot_info.initrd_base = @intFromPtr(image.ptr);
|
boot_information.initrd_base = @intFromPtr(image.ptr);
|
||||||
boot_info.initrd_len = image.len;
|
boot_information.initrd_len = image.len;
|
||||||
log("danos: initrd loaded\r\n");
|
log("danos: initrd loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
/// record each segment's layout so the kernel can re-map itself with the right
|
/// record each segment's layout so the kernel can re-map itself with the right
|
||||||
/// permissions.
|
/// permissions.
|
||||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize {
|
||||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||||
|
|
||||||
@@ -442,15 +442,15 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
// Record the virtual link address and the physical load address so the
|
// Record the virtual link address and the physical load address so the
|
||||||
// kernel can map itself with the right permissions post-switch. They're
|
// kernel can map itself with the right permissions post-switch. They're
|
||||||
// equal while the kernel links low; they diverge once it links high.
|
// equal while the kernel links low; they diverge once it links high.
|
||||||
const n = boot_info.kernel_segment_count;
|
const n = boot_information.kernel_segment_count;
|
||||||
if (n < boot_info.kernel_segments.len) {
|
if (n < boot_information.kernel_segments.len) {
|
||||||
boot_info.kernel_segments[n] = .{
|
boot_information.kernel_segments[n] = .{
|
||||||
.virt = phdr.p_vaddr,
|
.virtual = phdr.p_vaddr,
|
||||||
.phys = phdr.p_paddr,
|
.physical = phdr.p_paddr,
|
||||||
.pages = pages,
|
.pages = pages,
|
||||||
.flags = phdr.p_flags,
|
.flags = phdr.p_flags,
|
||||||
};
|
};
|
||||||
boot_info.kernel_segment_count = n + 1;
|
boot_information.kernel_segment_count = n + 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -467,21 +467,21 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
|||||||
const info = try bs.getMemoryMapInfo();
|
const info = try bs.getMemoryMapInfo();
|
||||||
// Spare descriptors to absorb the growth from the allocations below.
|
// Spare descriptors to absorb the growth from the allocations below.
|
||||||
const cap = info.len + 8;
|
const cap = info.len + 8;
|
||||||
const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||||
const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
||||||
const map = bs.getMemoryMap(map_buf) catch {
|
const map = bs.getMemoryMap(map_buffer) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
// Boot services are gone; do not touch `bs` again. Converting the map is
|
// Boot services are gone; do not touch `bs` again. Converting the map is
|
||||||
// pure computation on memory we already hold, so it's safe here.
|
// pure computation on memory we already hold, so it's safe here.
|
||||||
return convertMemoryMap(map, regions_buf);
|
return convertMemoryMap(map, regions_buffer);
|
||||||
}
|
}
|
||||||
return error.ExitBootServicesFailed;
|
return error.ExitBootServicesFailed;
|
||||||
}
|
}
|
||||||
@@ -513,11 +513,11 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
|
|
||||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||||
if (count > 0) {
|
if (count > 0) {
|
||||||
const prev = ®ions[count - 1];
|
const previous = ®ions[count - 1];
|
||||||
if (prev.kind == kind and
|
if (previous.kind == kind and
|
||||||
prev.base + prev.pages * danos.page_size == d.physical_start)
|
previous.base + previous.pages * danos.page_size == d.physical_start)
|
||||||
{
|
{
|
||||||
prev.pages += d.number_of_pages;
|
previous.pages += d.number_of_pages;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -533,7 +533,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
|
|
||||||
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
||||||
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
||||||
/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio`
|
/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio`
|
||||||
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
||||||
/// and such holes, and the cache attribute is what actually tells them apart.
|
/// and such holes, and the cache attribute is what actually tells them apart.
|
||||||
///
|
///
|
||||||
@@ -554,33 +554,33 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Write a compile-time string to the console (best effort).
|
/// Write a compile-time string to the console (best effort).
|
||||||
fn log(comptime msg: []const u8) void {
|
fn log(comptime message: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
var buf: [128]u16 = undefined;
|
var buffer: [128]u16 = undefined;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
for (bytes) |b| {
|
for (bytes) |b| {
|
||||||
if (i + 1 >= buf.len) break;
|
if (i + 1 >= buffer.len) break;
|
||||||
buf[i] = b;
|
buffer[i] = b;
|
||||||
i += 1;
|
i += 1;
|
||||||
}
|
}
|
||||||
buf[i] = 0;
|
buffer[i] = 0;
|
||||||
_ = out.outputString(buf[0..i :0].ptr) catch {};
|
_ = out.outputString(buffer[0..i :0].ptr) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
||||||
const table_entries = uefi.system_table.number_of_table_entries;
|
const table_entries = uefi.system_table.number_of_table_entries;
|
||||||
const config_tables = uefi.system_table.configuration_table;
|
const configuration_tables = uefi.system_table.configuration_table;
|
||||||
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
||||||
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
||||||
|
|
||||||
for (0..table_entries) |i| {
|
for (0..table_entries) |i| {
|
||||||
const entry = config_tables[i];
|
const entry = configuration_tables[i];
|
||||||
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
||||||
return entry.vendor_table;
|
return entry.vendor_table;
|
||||||
}
|
}
|
||||||
|
|||||||
+260
-222
@@ -11,43 +11,43 @@
|
|||||||
//!
|
//!
|
||||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||||
//! and is *not* mapped up front, so config-space pages are mapped on demand via
|
//! and is *not* mapped up front, so configuration-space pages are mapped on demand via
|
||||||
//! the `Hal.mapMmio` callback the caller supplies (the arch VMM's map primitive).
|
//! the `Hal.mapMmio` callback the caller supplies (the architecture VMM's map primitive).
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const aml = @import("aml/aml.zig");
|
const aml = @import("aml/aml.zig");
|
||||||
const DeviceTree = device.DeviceTree;
|
const DeviceTree = device_model.DeviceTree;
|
||||||
const Hal = device.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
/// A hardware register located either in MMIO or I/O-port space, as ACPI's
|
/// A hardware register located either in MMIO or I/O-port space, as ACPI's
|
||||||
/// Generic Address Structure describes. `address == 0` means "not present".
|
/// Generic Address Structure describes. `address == 0` means "not present".
|
||||||
pub const RegAccess = struct {
|
pub const RegisterAccess = struct {
|
||||||
/// true = system memory (MMIO), false = system I/O port space.
|
/// true = system memory (MMIO), false = system I/O port space.
|
||||||
mmio: bool = false,
|
mmio: bool = false,
|
||||||
address: u64 = 0,
|
address: u64 = 0,
|
||||||
/// Access width in bytes.
|
/// Access width in bytes.
|
||||||
width: u8 = 0,
|
width: u8 = 0,
|
||||||
|
|
||||||
pub fn present(self: RegAccess) bool {
|
pub fn present(self: RegisterAccess) bool {
|
||||||
return self.address != 0;
|
return self.address != 0;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||||
pub const PowerInfo = struct {
|
pub const PowerInformation = struct {
|
||||||
/// The SMM command port and the value that switches the platform into ACPI mode.
|
/// The SMM command port and the value that switches the platform into ACPI mode.
|
||||||
smi_cmd: u16 = 0,
|
smi_cmd: u16 = 0,
|
||||||
acpi_enable: u8 = 0,
|
acpi_enable: u8 = 0,
|
||||||
acpi_disable: u8 = 0,
|
acpi_disable: u8 = 0,
|
||||||
/// PM1 control registers — writing SLP_TYP|SLP_EN here enters a sleep state.
|
/// PM1 control registers — writing SLP_TYP|SLP_EN here enters a sleep state.
|
||||||
pm1a_cnt: RegAccess = .{},
|
pm1a_cnt: RegisterAccess = .{},
|
||||||
pm1b_cnt: RegAccess = .{},
|
pm1b_cnt: RegisterAccess = .{},
|
||||||
/// The FADT reset register and the value to write to it.
|
/// The FADT reset register and the value to write to it.
|
||||||
reset: RegAccess = .{},
|
reset: RegisterAccess = .{},
|
||||||
reset_value: u8 = 0,
|
reset_value: u8 = 0,
|
||||||
reset_supported: bool = false,
|
reset_supported: bool = false,
|
||||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`) packages.
|
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`) packages.
|
||||||
@@ -56,7 +56,7 @@ pub const PowerInfo = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||||
pub var power_info: PowerInfo = .{};
|
pub var power_information: PowerInformation = .{};
|
||||||
|
|
||||||
/// A legacy ISA IRQ remapped to a different global system interrupt (GSI), from a
|
/// A legacy ISA IRQ remapped to a different global system interrupt (GSI), from a
|
||||||
/// MADT Interrupt Source Override. `flags` are the MPS INTI polarity/trigger bits.
|
/// MADT Interrupt Source Override. `flags` are the MPS INTI polarity/trigger bits.
|
||||||
@@ -66,11 +66,11 @@ pub const IsoEntry = struct {
|
|||||||
flags: u16,
|
flags: u16,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Firmware facts the arch layer needs to avoid legacy assumptions (so danos boots
|
/// Firmware facts the architecture layer needs to avoid legacy assumptions (so danos boots
|
||||||
/// on legacy-free UEFI Class 3 machines). MMIO device *addresses* (HPET, IOAPIC)
|
/// on legacy-free UEFI Class 3 machines). MMIO device *addresses* (HPET, IOAPIC)
|
||||||
/// come from the device tree instead; this holds the scalar facts that have no
|
/// come from the device tree instead; this holds the scalar facts that have no
|
||||||
/// natural device node.
|
/// natural device node.
|
||||||
pub const PlatformInfo = struct {
|
pub const PlatformInformation = struct {
|
||||||
/// Whether the legacy 8259 PIC is present (MADT flags bit 0, PCAT_COMPAT). When
|
/// Whether the legacy 8259 PIC is present (MADT flags bit 0, PCAT_COMPAT). When
|
||||||
/// false, the PIC must not be programmed (it may not exist).
|
/// false, the PIC must not be programmed (it may not exist).
|
||||||
pic_present: bool = false,
|
pic_present: bool = false,
|
||||||
@@ -78,11 +78,11 @@ pub const PlatformInfo = struct {
|
|||||||
lapic_base: u64 = 0xFEE00000,
|
lapic_base: u64 = 0xFEE00000,
|
||||||
/// The ACPI power-management timer — a fixed 3.579545 MHz counter usable as a
|
/// The ACPI power-management timer — a fixed 3.579545 MHz counter usable as a
|
||||||
/// calibration reference when no HPET is present.
|
/// calibration reference when no HPET is present.
|
||||||
pm_timer: RegAccess = .{},
|
pm_timer: RegisterAccess = .{},
|
||||||
/// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VAL_EXT).
|
/// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VALUE_EXT).
|
||||||
pm_timer_32bit: bool = false,
|
pm_timer_32bit: bool = false,
|
||||||
/// The console UART the firmware points at (SPCR), if any — MMIO or I/O port.
|
/// The console UART the firmware points at (SPCR), if any — MMIO or I/O port.
|
||||||
spcr_uart: ?RegAccess = null,
|
spcr_uart: ?RegisterAccess = null,
|
||||||
/// SPCR interface type (0/1 = 16550/16450, …).
|
/// SPCR interface type (0/1 = 16550/16450, …).
|
||||||
spcr_kind: u8 = 0,
|
spcr_kind: u8 = 0,
|
||||||
/// ISA-IRQ-to-GSI remappings from the MADT (for future IOAPIC routing).
|
/// ISA-IRQ-to-GSI remappings from the MADT (for future IOAPIC routing).
|
||||||
@@ -90,8 +90,8 @@ pub const PlatformInfo = struct {
|
|||||||
override_count: usize = 0,
|
override_count: usize = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Filled in by `discover`; the arch layer reads it during bring-up.
|
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||||
pub var platform_info: PlatformInfo = .{};
|
pub var platform_information: PlatformInformation = .{};
|
||||||
|
|
||||||
/// One usable logical processor, from a MADT type-0 (Local APIC) record. The
|
/// One usable logical processor, from a MADT type-0 (Local APIC) record. The
|
||||||
/// `apic_id` is the Local APIC ID that SMP bring-up targets to wake this core
|
/// `apic_id` is the Local APIC ID that SMP bring-up targets to wake this core
|
||||||
@@ -108,19 +108,19 @@ pub const Cpu = struct {
|
|||||||
/// The set of usable logical processors the MADT listed — the hardware's degree of
|
/// The set of usable logical processors the MADT listed — the hardware's degree of
|
||||||
/// parallelism. Includes the bootstrap processor danos already runs on; the rest
|
/// parallelism. Includes the bootstrap processor danos already runs on; the rest
|
||||||
/// are the application processors SMP bring-up would start (see docs/smp.md).
|
/// are the application processors SMP bring-up would start (see docs/smp.md).
|
||||||
pub const CpuInfo = struct {
|
pub const CpuInformation = struct {
|
||||||
/// A static pool sized well above any danos target (a desktop, two 4-core Pis).
|
/// A static pool sized well above any danos target (a desktop, two 4-core Pis).
|
||||||
/// If the MADT ever lists more, the surplus is dropped and counted in `dropped`
|
/// If the MADT ever lists more, the surplus is dropped and counted in `dropped`
|
||||||
/// so the truncation is never silent.
|
/// so the truncation is never silent.
|
||||||
cpus: [max_cpus]Cpu = undefined,
|
cpus: [maximum_cpus]Cpu = undefined,
|
||||||
count: usize = 0,
|
count: usize = 0,
|
||||||
dropped: usize = 0,
|
dropped: usize = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const max_cpus = config.max_cpus;
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
|
|
||||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||||
pub var cpu_info: CpuInfo = .{};
|
pub var cpu_information: CpuInformation = .{};
|
||||||
|
|
||||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
||||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
/// walked every byte of the DSDT/SSDTs without desyncing.
|
||||||
@@ -136,20 +136,20 @@ pub var aml_stats: AmlStats = .{};
|
|||||||
pub var namespace: ?aml.Namespace = null;
|
pub var namespace: ?aml.Namespace = null;
|
||||||
|
|
||||||
/// Physical address of the DSDT the FADT points at, or 0.
|
/// Physical address of the DSDT the FADT points at, or 0.
|
||||||
pub var dsdt_phys: u64 = 0;
|
pub var dsdt_physical: u64 = 0;
|
||||||
|
|
||||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||||
// for the sleep-state (`_Sx`) packages.
|
// for the sleep-state (`_Sx`) packages.
|
||||||
var aml_block_phys: [32]u64 = undefined;
|
var aml_block_physical: [32]u64 = undefined;
|
||||||
var aml_block_len: [32]usize = undefined;
|
var aml_block_len: [32]usize = undefined;
|
||||||
var aml_block_count: usize = 0;
|
var aml_block_count: usize = 0;
|
||||||
|
|
||||||
fn addAmlBlock(sdt_phys: u64) void {
|
fn addAmlBlock(sdt_physical: u64) void {
|
||||||
if (aml_block_count >= aml_block_phys.len or sdt_phys == 0) return;
|
if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return;
|
||||||
const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys));
|
const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(sdt_physical));
|
||||||
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
||||||
aml_block_phys[aml_block_count] = sdt_phys + @sizeOf(SystemDescriptorTableHeader);
|
aml_block_physical[aml_block_count] = sdt_physical + @sizeOf(SystemDescriptorTableHeader);
|
||||||
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
||||||
aml_block_count += 1;
|
aml_block_count += 1;
|
||||||
}
|
}
|
||||||
@@ -362,47 +362,47 @@ const PciHeader = extern struct {
|
|||||||
|
|
||||||
// --- Entry point ------------------------------------------------------------
|
// --- Entry point ------------------------------------------------------------
|
||||||
|
|
||||||
/// Discover hardware from the ACPI tables rooted at `rsdp_phys` and populate
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
/// `dt`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_info` for the power service.
|
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||||
pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
if (rsdp_phys == 0) return error.NoRsdp;
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
|
|
||||||
// Start clean so a re-run doesn't accumulate stale state.
|
// Start clean so a re-run doesn't accumulate stale state.
|
||||||
power_info = .{};
|
power_information = .{};
|
||||||
platform_info = .{};
|
platform_information = .{};
|
||||||
aml_stats = .{};
|
aml_stats = .{};
|
||||||
namespace = null;
|
namespace = null;
|
||||||
dsdt_phys = 0;
|
dsdt_physical = 0;
|
||||||
aml_block_count = 0;
|
aml_block_count = 0;
|
||||||
|
|
||||||
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physToVirt(rsdp_phys));
|
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physicalToVirtual(rsdp_physical));
|
||||||
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
||||||
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
||||||
if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), 20)) return error.BadRsdpChecksum;
|
if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum;
|
||||||
|
|
||||||
if (rsdp.revision >= 2) {
|
if (rsdp.revision >= 2) {
|
||||||
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physToVirt(rsdp_phys));
|
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physicalToVirtual(rsdp_physical));
|
||||||
if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), xsdp.length)) return error.BadXsdpChecksum;
|
if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum;
|
||||||
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, dt, hal);
|
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, device_tree, hal);
|
||||||
} else {
|
} else {
|
||||||
try walkRoot(u32, rsdp.root_system_description_table_address, dt, hal);
|
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
||||||
// read the sleep types from it.
|
// read the sleep types from it.
|
||||||
var blocks: [aml_block_phys.len][]const u8 = undefined;
|
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||||
for (0..aml_block_count) |i| {
|
for (0..aml_block_count) |i| {
|
||||||
blocks[i] = @as([*]const u8, @ptrFromInt(danos.physToVirt(aml_block_phys[i])))[0..aml_block_len[i]];
|
blocks[i] = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||||
}
|
}
|
||||||
const active = blocks[0..aml_block_count];
|
const active = blocks[0..aml_block_count];
|
||||||
if (aml.parse(dt.allocator, active)) |pr| {
|
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||||
namespace = pr.namespace;
|
namespace = pr.namespace;
|
||||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||||
power_info.s5 = aml.sleepState(&namespace.?, 5);
|
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||||
power_info.s3 = aml.sleepState(&namespace.?, 3);
|
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||||
// Fold the namespace's Device objects into the generic tree.
|
// Fold the namespace's Device objects into the generic tree.
|
||||||
wireAcpiDevices(dt, &namespace.?, hal) catch {};
|
wireAcpiDevices(device_tree, &namespace.?, hal) catch {};
|
||||||
} else |_| {
|
} else |_| {
|
||||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||||
// whatever the FADT alone provided.
|
// whatever the FADT alone provided.
|
||||||
@@ -411,51 +411,51 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
|||||||
|
|
||||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(root_phys));
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(root_physical));
|
||||||
if (!checksumOk(@ptrFromInt(danos.physToVirt(root_phys)), header.length)) return error.BadRootChecksum;
|
if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum;
|
||||||
|
|
||||||
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
||||||
const base: [*]const u8 = @ptrFromInt(danos.physToVirt(root_phys));
|
const base: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(root_physical));
|
||||||
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
||||||
|
|
||||||
for (entries[0..count]) |ent| {
|
for (entries[0..count]) |ent| {
|
||||||
const sdt_phys: u64 = ent; // u32 entries widen; u64 pass through
|
const sdt_physical: u64 = ent; // u32 entries widen; u64 pass through
|
||||||
handleTable(dt, hal, sdt_phys) catch continue;
|
handleTable(device_tree, hal, sdt_physical) catch continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Dispatch a single SDT on its signature.
|
/// Dispatch a single SDT on its signature.
|
||||||
fn handleTable(dt: *DeviceTree, hal: Hal, sdt_phys: u64) !void {
|
fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys));
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(sdt_physical));
|
||||||
const sig = header.signature;
|
const sig = header.signature;
|
||||||
if (std.mem.eql(u8, &sig, &APIC)) {
|
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||||
try parseMadt(dt, header);
|
try parseMadt(device_tree, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
||||||
try parseMcfg(dt, hal, header);
|
try parseMcfg(device_tree, hal, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
||||||
try parseHpet(dt, header);
|
try parseHpet(device_tree, hal, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
||||||
parseFadt(header);
|
parseFadt(header);
|
||||||
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||||
parseSpcr(header);
|
parseSpcr(header);
|
||||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
||||||
addAmlBlock(sdt_phys);
|
addAmlBlock(sdt_physical);
|
||||||
}
|
}
|
||||||
// Any other signature is recognised but left opaque for now.
|
// Any other signature is recognised but left opaque for now.
|
||||||
}
|
}
|
||||||
|
|
||||||
/// MADT -> one processor node per Local APIC, one interrupt_controller per I/O APIC.
|
/// MADT -> one processor node per Local APIC, one interrupt_controller per I/O APIC.
|
||||||
fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||||
const madt: *const Madt = @ptrCast(header);
|
const madt: *const Madt = @ptrCast(header);
|
||||||
const total: usize = header.length;
|
const total: usize = header.length;
|
||||||
const base: [*]const u8 = @ptrCast(header);
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
var ioapic_index: usize = 0;
|
var ioapic_index: usize = 0;
|
||||||
|
|
||||||
// MADT header: local APIC base + flags (bit 0 = 8259 PIC present).
|
// MADT header: local APIC base + flags (bit 0 = 8259 PIC present).
|
||||||
platform_info.lapic_base = madt.local_apic_address;
|
platform_information.lapic_base = madt.local_apic_address;
|
||||||
platform_info.pic_present = madt.flags & 1 != 0;
|
platform_information.pic_present = madt.flags & 1 != 0;
|
||||||
|
|
||||||
var off: usize = @sizeOf(Madt);
|
var off: usize = @sizeOf(Madt);
|
||||||
while (off + @sizeOf(MadtRecordHeader) <= total) {
|
while (off + @sizeOf(MadtRecordHeader) <= total) {
|
||||||
@@ -468,18 +468,18 @@ fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void
|
|||||||
if (la.flags & 1 != 0) {
|
if (la.flags & 1 != 0) {
|
||||||
var nb: [24]u8 = undefined;
|
var nb: [24]u8 = undefined;
|
||||||
const nm = std.fmt.bufPrint(&nb, "cpu{d}", .{la.processor_id}) catch "cpu";
|
const nm = std.fmt.bufPrint(&nb, "cpu{d}", .{la.processor_id}) catch "cpu";
|
||||||
_ = try dt.addChild(dt.root, .processor, nm);
|
_ = try device_tree.addChild(device_tree.root, .processor, nm);
|
||||||
// Also record it as a schedulable core (with the APIC ID an AP
|
// Also record it as a schedulable core (with the APIC ID an AP
|
||||||
// wake needs, which the device node name doesn't preserve).
|
// wake needs, which the device node name doesn't preserve).
|
||||||
if (cpu_info.count < cpu_info.cpus.len) {
|
if (cpu_information.count < cpu_information.cpus.len) {
|
||||||
cpu_info.cpus[cpu_info.count] = .{
|
cpu_information.cpus[cpu_information.count] = .{
|
||||||
.processor_id = la.processor_id,
|
.processor_id = la.processor_id,
|
||||||
.apic_id = la.apic_id,
|
.apic_id = la.apic_id,
|
||||||
.online_capable = la.flags & 2 != 0,
|
.online_capable = la.flags & 2 != 0,
|
||||||
};
|
};
|
||||||
cpu_info.count += 1;
|
cpu_information.count += 1;
|
||||||
} else {
|
} else {
|
||||||
cpu_info.dropped += 1;
|
cpu_information.dropped += 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
@@ -488,25 +488,25 @@ fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void
|
|||||||
var nb: [24]u8 = undefined;
|
var nb: [24]u8 = undefined;
|
||||||
const nm = std.fmt.bufPrint(&nb, "ioapic{d}", .{ioapic_index}) catch "ioapic";
|
const nm = std.fmt.bufPrint(&nb, "ioapic{d}", .{ioapic_index}) catch "ioapic";
|
||||||
ioapic_index += 1;
|
ioapic_index += 1;
|
||||||
const d = try dt.addChild(dt.root, .interrupt_controller, nm);
|
const d = try device_tree.addChild(device_tree.root, .interrupt_controller, nm);
|
||||||
_ = d.addResource(.memory, io.address, 0x20);
|
_ = d.addResource(.memory, io.address, 0x20);
|
||||||
// The GSI range this I/O APIC handles, starting at gsi_base.
|
// The GSI range this I/O APIC handles, starting at gsi_base.
|
||||||
_ = d.addResource(.irq, io.gsi_base, 0);
|
_ = d.addResource(.irq, io.gsi_base, 0);
|
||||||
},
|
},
|
||||||
2 => {
|
2 => {
|
||||||
const iso: *const MadtIso = @ptrCast(base + off);
|
const iso: *const MadtIso = @ptrCast(base + off);
|
||||||
if (platform_info.override_count < platform_info.overrides.len) {
|
if (platform_information.override_count < platform_information.overrides.len) {
|
||||||
platform_info.overrides[platform_info.override_count] = .{
|
platform_information.overrides[platform_information.override_count] = .{
|
||||||
.source = iso.source,
|
.source = iso.source,
|
||||||
.gsi = iso.gsi,
|
.gsi = iso.gsi,
|
||||||
.flags = iso.flags,
|
.flags = iso.flags,
|
||||||
};
|
};
|
||||||
platform_info.override_count += 1;
|
platform_information.override_count += 1;
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
5 => {
|
5 => {
|
||||||
const ovr: *const MadtLapicOverride = @ptrCast(base + off);
|
const ovr: *const MadtLapicOverride = @ptrCast(base + off);
|
||||||
platform_info.lapic_base = ovr.address;
|
platform_information.lapic_base = ovr.address;
|
||||||
},
|
},
|
||||||
else => {},
|
else => {},
|
||||||
}
|
}
|
||||||
@@ -515,7 +515,7 @@ fn parseMadt(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
||||||
fn parseMcfg(dt: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
fn parseMcfg(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
||||||
const total: usize = header.length;
|
const total: usize = header.length;
|
||||||
const base: [*]const u8 = @ptrCast(header);
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
|
|
||||||
@@ -526,12 +526,12 @@ fn parseMcfg(dt: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHead
|
|||||||
|
|
||||||
var nb: [24]u8 = undefined;
|
var nb: [24]u8 = undefined;
|
||||||
const nm = std.fmt.bufPrint(&nb, "pci{d}", .{alloc.segment_group}) catch "pci";
|
const nm = std.fmt.bufPrint(&nb, "pci{d}", .{alloc.segment_group}) catch "pci";
|
||||||
const bridge = try dt.addChild(dt.root, .pci_host_bridge, nm);
|
const bridge = try device_tree.addChild(device_tree.root, .pci_host_bridge, nm);
|
||||||
// ECAM window: 1 MiB of config space per bus.
|
// ECAM window: 1 MiB of configuration space per bus.
|
||||||
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
||||||
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
||||||
|
|
||||||
try enumeratePci(dt, bridge, hal, alloc.*);
|
try enumeratePci(device_tree, bridge, hal, alloc.*);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -539,38 +539,38 @@ fn parseMcfg(dt: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHead
|
|||||||
/// bridge recursion yet: on the ECAM path the host bridge decodes every bus in
|
/// bridge recursion yet: on the ECAM path the host bridge decodes every bus in
|
||||||
/// the window, so scanning the declared range finds everything QEMU exposes.
|
/// the window, so scanning the declared range finds everything QEMU exposes.
|
||||||
fn enumeratePci(
|
fn enumeratePci(
|
||||||
dt: *DeviceTree,
|
device_tree: *DeviceTree,
|
||||||
bridge: *device.Device,
|
bridge: *device_model.Device,
|
||||||
hal: Hal,
|
hal: Hal,
|
||||||
alloc: McfgAllocation,
|
alloc: McfgAllocation,
|
||||||
) !void {
|
) !void {
|
||||||
var bus: u16 = alloc.start_bus;
|
var bus: u16 = alloc.start_bus;
|
||||||
while (bus <= alloc.end_bus) : (bus += 1) {
|
while (bus <= alloc.end_bus) : (bus += 1) {
|
||||||
var dev: u8 = 0;
|
var device: u8 = 0;
|
||||||
while (dev < 32) : (dev += 1) {
|
while (device < 32) : (device += 1) {
|
||||||
const h0: *align(1) const PciHeader = @ptrCast(pciConfigPtr(alloc, hal, @intCast(bus), dev, 0));
|
const h0: *align(1) const PciHeader = @ptrCast(pciConfigurationPtr(alloc, hal, @intCast(bus), device, 0));
|
||||||
if (h0.vendor_id == 0xFFFF) continue; // no function 0 => slot empty
|
if (h0.vendor_id == 0xFFFF) continue; // no function 0 => slot empty
|
||||||
|
|
||||||
const funcs: u8 = if (h0.header_type & 0x80 != 0) 8 else 1;
|
const funcs: u8 = if (h0.header_type & 0x80 != 0) 8 else 1;
|
||||||
var func: u8 = 0;
|
var function: u8 = 0;
|
||||||
while (func < funcs) : (func += 1) {
|
while (function < funcs) : (function += 1) {
|
||||||
const cfg = pciConfigPtr(alloc, hal, @intCast(bus), dev, func);
|
const configuration = pciConfigurationPtr(alloc, hal, @intCast(bus), device, function);
|
||||||
const h: *align(1) const PciHeader = @ptrCast(cfg);
|
const h: *align(1) const PciHeader = @ptrCast(configuration);
|
||||||
if (h.vendor_id == 0xFFFF) continue;
|
if (h.vendor_id == 0xFFFF) continue;
|
||||||
|
|
||||||
var nb: [24]u8 = undefined;
|
var nb: [24]u8 = undefined;
|
||||||
const nm = std.fmt.bufPrint(&nb, "{s}:{x:0>2}:{x:0>2}.{d}", .{
|
const nm = std.fmt.bufPrint(&nb, "{s}:{x:0>2}:{x:0>2}.{d}", .{
|
||||||
bridge.name(), bus, dev, func,
|
bridge.name(), bus, device, function,
|
||||||
}) catch "pcidev";
|
}) catch "pcidev";
|
||||||
const node = try dt.addChild(bridge, .pci_device, nm);
|
const node = try device_tree.addChild(bridge, .pci_device, nm);
|
||||||
node.ids.pci_vendor = h.vendor_id;
|
node.ids.pci_vendor = h.vendor_id;
|
||||||
node.ids.pci_device = h.device_id;
|
node.ids.pci_device = h.device_id;
|
||||||
node.ids.pci_class = (@as(u24, h.class_code) << 16) |
|
node.ids.pci_class = (@as(u24, h.class_code) << 16) |
|
||||||
(@as(u24, h.subclass) << 8) | h.prog_if;
|
(@as(u24, h.subclass) << 8) | h.prog_if;
|
||||||
node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, dev) << 3) | func;
|
node.ids.pci_bdf = (@as(u16, @intCast(bus)) << 8) | (@as(u16, device) << 3) | function;
|
||||||
|
|
||||||
// BARs only exist in header type 0 (normal devices), not bridges.
|
// BARs only exist in header type 0 (normal devices), not bridges.
|
||||||
if (h.header_type & 0x7F == 0) addBars(node, cfg);
|
if (h.header_type & 0x7F == 0) addBars(node, configuration);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -579,58 +579,96 @@ fn enumeratePci(
|
|||||||
/// Record and size the memory/IO windows named by a device's Base Address
|
/// Record and size the memory/IO windows named by a device's Base Address
|
||||||
/// Registers. Sizing is the standard probe: disable decode, write all-ones, read
|
/// Registers. Sizing is the standard probe: disable decode, write all-ones, read
|
||||||
/// back the writable (address) bits, restore. `size = ~mask + 1`.
|
/// back the writable (address) bits, restore. `size = ~mask + 1`.
|
||||||
fn addBars(node: *device.Device, cfg: [*]align(1) u8) void {
|
fn addBars(node: *device_model.Device, configuration: [*]align(1) u8) void {
|
||||||
// Stop the device decoding its BARs while we transiently write all-ones.
|
// Stop the device decoding its BARs while we transiently write all-ones.
|
||||||
const command = rd(u16, cfg, 0x04);
|
const command = rd(u16, configuration, 0x04);
|
||||||
wr(u16, cfg, 0x04, command & ~@as(u16, 0b11));
|
wr(u16, configuration, 0x04, command & ~@as(u16, 0b11));
|
||||||
|
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < 6) : (i += 1) {
|
while (i < 6) : (i += 1) {
|
||||||
const off = 0x10 + i * 4;
|
const off = 0x10 + i * 4;
|
||||||
const orig = rd(u32, cfg, off);
|
const orig = rd(u32, configuration, off);
|
||||||
if (orig == 0) continue;
|
if (orig == 0) continue;
|
||||||
|
|
||||||
if (orig & 1 != 0) {
|
if (orig & 1 != 0) {
|
||||||
// I/O-space BAR (16-bit address space on x86).
|
// I/O-space BAR (16-bit address space on x86).
|
||||||
wr(u32, cfg, off, 0xFFFF_FFFF);
|
wr(u32, configuration, off, 0xFFFF_FFFF);
|
||||||
const readback = rd(u32, cfg, off);
|
const readback = rd(u32, configuration, off);
|
||||||
wr(u32, cfg, off, orig);
|
wr(u32, configuration, off, orig);
|
||||||
const mask = readback & 0xFFFF_FFFC;
|
const mask = readback & 0xFFFF_FFFC;
|
||||||
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
|
const size: u32 = if (mask == 0) 0 else (~mask +% 1) & 0xFFFF;
|
||||||
_ = node.addResource(.io_port, orig & 0xFFFF_FFFC, size);
|
_ = node.addResource(.io_port, orig & 0xFFFF_FFFC, size);
|
||||||
} else if ((orig >> 1) & 0x3 == 2) {
|
} else if ((orig >> 1) & 0x3 == 2) {
|
||||||
// 64-bit memory BAR: this BAR pair spans two config slots.
|
// 64-bit memory BAR: this BAR pair spans two configuration slots.
|
||||||
const orig_hi = rd(u32, cfg, off + 4);
|
const orig_hi = rd(u32, configuration, off + 4);
|
||||||
wr(u32, cfg, off, 0xFFFF_FFFF);
|
wr(u32, configuration, off, 0xFFFF_FFFF);
|
||||||
wr(u32, cfg, off + 4, 0xFFFF_FFFF);
|
wr(u32, configuration, off + 4, 0xFFFF_FFFF);
|
||||||
const lo = rd(u32, cfg, off);
|
const lo = rd(u32, configuration, off);
|
||||||
const hi = rd(u32, cfg, off + 4);
|
const hi = rd(u32, configuration, off + 4);
|
||||||
wr(u32, cfg, off, orig);
|
wr(u32, configuration, off, orig);
|
||||||
wr(u32, cfg, off + 4, orig_hi);
|
wr(u32, configuration, off + 4, orig_hi);
|
||||||
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
|
const readback = (@as(u64, hi) << 32) | (lo & 0xFFFF_FFF0);
|
||||||
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
|
const size: u64 = if (readback == 0) 0 else ~readback +% 1;
|
||||||
const addr = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0);
|
const address = (@as(u64, orig_hi) << 32) | (orig & 0xFFFF_FFF0);
|
||||||
_ = node.addResource(.memory, addr, size);
|
_ = node.addResource(.memory, address, size);
|
||||||
i += 1; // consumed the high half
|
i += 1; // consumed the high half
|
||||||
} else {
|
} else {
|
||||||
// 32-bit memory BAR.
|
// 32-bit memory BAR.
|
||||||
wr(u32, cfg, off, 0xFFFF_FFFF);
|
wr(u32, configuration, off, 0xFFFF_FFFF);
|
||||||
const readback = rd(u32, cfg, off);
|
const readback = rd(u32, configuration, off);
|
||||||
wr(u32, cfg, off, orig);
|
wr(u32, configuration, off, orig);
|
||||||
const mask = readback & 0xFFFF_FFF0;
|
const mask = readback & 0xFFFF_FFF0;
|
||||||
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
|
const size: u32 = if (mask == 0) 0 else ~mask +% 1;
|
||||||
_ = node.addResource(.memory, orig & 0xFFFF_FFF0, size);
|
_ = node.addResource(.memory, orig & 0xFFFF_FFF0, size);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
wr(u16, cfg, 0x04, command); // restore decode
|
wr(u16, configuration, 0x04, command); // restore decode
|
||||||
}
|
}
|
||||||
|
|
||||||
/// HPET -> a timer node with its register block as an MMIO resource.
|
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
|
||||||
fn parseHpet(dt: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
/// its comparators can raise.
|
||||||
|
///
|
||||||
|
/// Unlike a PCI device or an ACPI `_CRS` node, the HPET table carries **no interrupt
|
||||||
|
/// number**: which I/O APIC inputs a comparator may drive is advertised at runtime,
|
||||||
|
/// as a bitmask in `Tn_INT_ROUTE_CAP` (bits 63:32 of the Timer 0 configuration register).
|
||||||
|
/// So discovery maps the register block, reads the mask, and records one concrete
|
||||||
|
/// `irq` resource — the GSI a driver is entitled to bind. The driver commits to it
|
||||||
|
/// by writing `Tn_INT_ROUTE_CNF`; the kernel checks the binding against this
|
||||||
|
/// resource (see process.ownedGsi), which is what keeps `irq_bind` a capability
|
||||||
|
/// rather than a request for an arbitrary interrupt line.
|
||||||
|
fn parseHpet(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
||||||
const hpet: *const Hpet = @ptrCast(header);
|
const hpet: *const Hpet = @ptrCast(header);
|
||||||
const d = try dt.addChild(dt.root, .timer, "hpet");
|
const d = try device_tree.addChild(device_tree.root, .timer, "hpet");
|
||||||
|
|
||||||
|
// The GAS tag must say System Memory (0) before we treat `address` as a physical
|
||||||
|
// address. The HPET spec mandates it, but firmware is not a thing to trust: a
|
||||||
|
// System I/O (1) tag here would have us map an arbitrary page and read a bogus
|
||||||
|
// route-capability mask out of it.
|
||||||
|
if (hpet.address_space_id != gas_system_memory) return;
|
||||||
|
|
||||||
_ = d.addResource(.memory, hpet.address, 0x400);
|
_ = d.addResource(.memory, hpet.address, 0x400);
|
||||||
|
|
||||||
|
const regs = hal.mapMmio(hpet.address, 0x400, true);
|
||||||
|
const t0_configuration: *const volatile u64 = @ptrFromInt(regs + 0x100);
|
||||||
|
const route_cap: u32 = @truncate(t0_configuration.* >> 32);
|
||||||
|
if (hpetGsi(route_cap)) |gsi| _ = d.addResource(.irq, gsi, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ACPI Generic Address Structure address-space ids we care about.
|
||||||
|
const gas_system_memory: u8 = 0;
|
||||||
|
|
||||||
|
/// Pick a GSI for the HPET out of its route-capability mask. Prefer an input at or
|
||||||
|
/// above 16: the low ones overlap the legacy ISA lines (2 = cascaded PIT, 8 = RTC),
|
||||||
|
/// which the MADT may separately override, whereas 16+ are the free upper inputs on
|
||||||
|
/// every I/O APIC we care about. Falls back to the lowest bit set if there are none.
|
||||||
|
fn hpetGsi(route_cap: u32) ?u32 {
|
||||||
|
if (route_cap == 0) return null;
|
||||||
|
var gsi: u32 = 16;
|
||||||
|
while (gsi < 32) : (gsi += 1) {
|
||||||
|
if (route_cap & (@as(u32, 1) << @intCast(gsi)) != 0) return gsi;
|
||||||
|
}
|
||||||
|
return @ctz(route_cap);
|
||||||
}
|
}
|
||||||
|
|
||||||
// FADT field offsets (bytes from the table start). The FADT grew across ACPI
|
// FADT field offsets (bytes from the table start). The FADT grew across ACPI
|
||||||
@@ -645,45 +683,45 @@ const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
|
|||||||
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
||||||
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
||||||
const fadt_flags = 112; // u32
|
const fadt_flags = 112; // u32
|
||||||
const fadt_reset_reg = 116; // GAS (12 bytes)
|
const fadt_reset_register = 116; // GAS (12 bytes)
|
||||||
const fadt_reset_value = 128; // u8
|
const fadt_reset_value = 128; // u8
|
||||||
const fadt_x_dsdt = 140; // u64
|
const fadt_x_dsdt = 140; // u64
|
||||||
const fadt_x_pm1a_cnt_blk = 172; // GAS
|
const fadt_x_pm1a_cnt_blk = 172; // GAS
|
||||||
const fadt_x_pm1b_cnt_blk = 184; // GAS
|
const fadt_x_pm1b_cnt_blk = 184; // GAS
|
||||||
const fadt_x_pm_tmr_blk = 208; // GAS
|
const fadt_x_pm_tmr_blk = 208; // GAS
|
||||||
const flag_reset_reg_supported = 1 << 10;
|
const flag_reset_register_supported = 1 << 10;
|
||||||
const flag_tmr_val_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||||
|
|
||||||
/// FADT -> the power register map (into `power_info`) and the DSDT address, which
|
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
||||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
||||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
const len: usize = header.length;
|
const len: usize = header.length;
|
||||||
const pi = &power_info;
|
const pi = &power_information;
|
||||||
|
|
||||||
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
||||||
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
||||||
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
||||||
|
|
||||||
const cnt_width = fadt(u8, base, len, fadt_pm1_cnt_len) orelse 2;
|
const cnt_width = fadt(u8, base, len, fadt_pm1_cnt_len) orelse 2;
|
||||||
pi.pm1a_cnt = readCntReg(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width);
|
pi.pm1a_cnt = readCntRegister(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width);
|
||||||
pi.pm1b_cnt = readCntReg(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width);
|
pi.pm1b_cnt = readCntRegister(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width);
|
||||||
|
|
||||||
const flags = fadt(u32, base, len, fadt_flags) orelse 0;
|
const flags = fadt(u32, base, len, fadt_flags) orelse 0;
|
||||||
pi.reset_supported = flags & flag_reset_reg_supported != 0;
|
pi.reset_supported = flags & flag_reset_register_supported != 0;
|
||||||
pi.reset = readGas(base, len, fadt_reset_reg) orelse .{};
|
pi.reset = readGas(base, len, fadt_reset_register) orelse .{};
|
||||||
pi.reset_value = fadt(u8, base, len, fadt_reset_value) orelse 0;
|
pi.reset_value = fadt(u8, base, len, fadt_reset_value) orelse 0;
|
||||||
|
|
||||||
// The PM timer — a fixed-rate counter used as a calibration reference when no
|
// The PM timer — a fixed-rate counter used as a calibration reference when no
|
||||||
// HPET is present. Prefer the 64-bit-capable X_ GAS, fall back to the port.
|
// HPET is present. Prefer the 64-bit-capable X_ GAS, fall back to the port.
|
||||||
platform_info.pm_timer = readCntReg(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4);
|
platform_information.pm_timer = readCntRegister(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4);
|
||||||
platform_info.pm_timer_32bit = flags & flag_tmr_val_ext != 0;
|
platform_information.pm_timer_32bit = flags & flag_tmr_value_ext != 0;
|
||||||
|
|
||||||
var dsdt: u64 = fadt(u32, base, len, fadt_dsdt) orelse 0;
|
var dsdt: u64 = fadt(u32, base, len, fadt_dsdt) orelse 0;
|
||||||
if (fadt(u64, base, len, fadt_x_dsdt)) |x| {
|
if (fadt(u64, base, len, fadt_x_dsdt)) |x| {
|
||||||
if (x != 0) dsdt = x;
|
if (x != 0) dsdt = x;
|
||||||
}
|
}
|
||||||
dsdt_phys = dsdt;
|
dsdt_physical = dsdt;
|
||||||
addAmlBlock(dsdt);
|
addAmlBlock(dsdt);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -698,15 +736,15 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
|||||||
const len: usize = header.length;
|
const len: usize = header.length;
|
||||||
const gas = readGas(base, len, spcr_base_address) orelse return;
|
const gas = readGas(base, len, spcr_base_address) orelse return;
|
||||||
if (gas.address == 0) return;
|
if (gas.address == 0) return;
|
||||||
platform_info.spcr_uart = gas;
|
platform_information.spcr_uart = gas;
|
||||||
platform_info.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0;
|
platform_information.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- AML namespace -> generic device tree -----------------------------------
|
// --- AML namespace -> generic device tree -----------------------------------
|
||||||
|
|
||||||
/// The PCI bus context while descending the ACPI namespace: the generic host
|
/// The PCI bus context while descending the ACPI namespace: the generic host
|
||||||
/// bridge whose children ACPI address (`_ADR`) devices resolve against, and the bus number.
|
/// bridge whose children ACPI address (`_ADR`) devices resolve against, and the bus number.
|
||||||
const PciCtx = struct { bridge: *device.Device, bus: u8 };
|
const PciContext = struct { bridge: *device_model.Device, bus: u8 };
|
||||||
|
|
||||||
/// Mirror the ACPI namespace's Device objects into the generic tree, *merging*
|
/// Mirror the ACPI namespace's Device objects into the generic tree, *merging*
|
||||||
/// them with the PCI-enumerated nodes: a PCI root bridge (`PNP0A03`/`PNP0A08`)
|
/// them with the PCI-enumerated nodes: a PCI root bridge (`PNP0A03`/`PNP0A08`)
|
||||||
@@ -714,73 +752,73 @@ const PciCtx = struct { bridge: *device.Device, bus: u8 };
|
|||||||
/// the matching PCI function (annotating it with the ACPI hardware ID (`_HID`) and nesting the
|
/// the matching PCI function (annotating it with the ACPI hardware ID (`_HID`) and nesting the
|
||||||
/// ACPI-only children — keyboard, RTC, … — beneath it). Namespace devices with no
|
/// ACPI-only children — keyboard, RTC, … — beneath it). Namespace devices with no
|
||||||
/// PCI match land under a synthetic `acpi` node.
|
/// PCI match land under a synthetic `acpi` node.
|
||||||
fn wireAcpiDevices(dt: *DeviceTree, nsp: *aml.Namespace, hal: Hal) !void {
|
fn wireAcpiDevices(device_tree: *DeviceTree, aml_namespace: *aml.Namespace, hal: Hal) !void {
|
||||||
var arena = std.heap.ArenaAllocator.init(dt.allocator);
|
var arena = std.heap.ArenaAllocator.init(device_tree.allocator);
|
||||||
defer arena.deinit();
|
defer arena.deinit();
|
||||||
var ev = aml.Interp.init(nsp, .{
|
var interpreter = aml.Interpreter.init(aml_namespace, .{
|
||||||
.mapMmio = hal.mapMmio,
|
.mapMmio = hal.mapMmio,
|
||||||
.pioRead = hal.pioRead,
|
.pioRead = hal.pioRead,
|
||||||
.pioWrite = hal.pioWrite,
|
.pioWrite = hal.pioWrite,
|
||||||
}, arena.allocator());
|
}, arena.allocator());
|
||||||
|
|
||||||
const acpi_root = try dt.addChild(dt.root, .unknown, "acpi");
|
const acpi_root = try device_tree.addChild(device_tree.root, .unknown, "acpi");
|
||||||
try mirrorDevices(dt, nsp.root, acpi_root, null, &ev);
|
try mirrorDevices(device_tree, aml_namespace.root, acpi_root, null, &interpreter);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn mirrorDevices(dt: *DeviceTree, node: *aml.Node, parent_dev: *device.Device, ctx: ?PciCtx, ev: *aml.Interp) (error{OutOfMemory})!void {
|
fn mirrorDevices(device_tree: *DeviceTree, node: *aml.Node, parent_device: *device_model.Device, context: ?PciContext, interpreter: *aml.Interpreter) (error{OutOfMemory})!void {
|
||||||
var child = node.first_child;
|
var child = node.first_child;
|
||||||
while (child) |c| : (child = c.next_sibling) {
|
while (child) |c| : (child = c.next_sibling) {
|
||||||
if (c.kind != .device) {
|
if (c.kind != .device) {
|
||||||
// A scope — the System Bus (\_SB), General Purpose Events (\_GPE), … —
|
// A scope — the System Bus (\_SB), General Purpose Events (\_GPE), … —
|
||||||
// descend without adding a node.
|
// descend without adding a node.
|
||||||
try mirrorDevices(dt, c, parent_dev, ctx, ev);
|
try mirrorDevices(device_tree, c, parent_device, context, interpreter);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Skip devices the firmware reports as not present (via a device-status (`_STA`) method),
|
// Skip devices the firmware reports as not present (via a device-status (`_STA`) method),
|
||||||
// along with their whole subtree — per the ACPI rules.
|
// along with their whole subtree — per the ACPI rules.
|
||||||
if (!devicePresent(ev, c)) continue;
|
if (!devicePresent(interpreter, c)) continue;
|
||||||
|
|
||||||
var gdev: *device.Device = undefined;
|
var mirrored_device: *device_model.Device = undefined;
|
||||||
var child_ctx = ctx;
|
var child_context = context;
|
||||||
|
|
||||||
if (isPciRootNode(c)) {
|
if (isPciRootNode(c)) {
|
||||||
// The PCI root bridge folds onto the generic host bridge.
|
// The PCI root bridge folds onto the generic host bridge.
|
||||||
gdev = matchHostBridge(dt) orelse
|
mirrored_device = matchHostBridge(device_tree) orelse
|
||||||
try dt.addChild(parent_dev, .acpi_device, &c.seg);
|
try device_tree.addChild(parent_device, .acpi_device, &c.segment);
|
||||||
child_ctx = .{ .bridge = gdev, .bus = 0 };
|
child_context = .{ .bridge = mirrored_device, .bus = 0 };
|
||||||
} else {
|
} else {
|
||||||
// An addressed device folds onto its matching PCI function; anything
|
// An addressed device folds onto its matching PCI function; anything
|
||||||
// else becomes a fresh node under the current parent.
|
// else becomes a fresh node under the current parent.
|
||||||
gdev = pick: {
|
mirrored_device = pick: {
|
||||||
if (ctx) |pc| {
|
if (context) |pc| {
|
||||||
if (readAdr(c)) |adr| {
|
if (readAdr(c)) |adr| {
|
||||||
if (findPciNode(pc.bridge, pc.bus, adr)) |pnode| break :pick pnode;
|
if (findPciNode(pc.bridge, pc.bus, adr)) |pnode| break :pick pnode;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
break :pick try dt.addChild(parent_dev, .acpi_device, &c.seg);
|
break :pick try device_tree.addChild(parent_device, .acpi_device, &c.segment);
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
applyHid(gdev, c, ev);
|
applyHid(mirrored_device, c, interpreter);
|
||||||
applyCrs(gdev, c, ev);
|
applyCrs(mirrored_device, c, interpreter);
|
||||||
try mirrorDevices(dt, c, gdev, child_ctx, ev);
|
try mirrorDevices(device_tree, c, mirrored_device, child_context, interpreter);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Evaluate a device's status (`_STA`) to decide if it is present. An absent status
|
/// Evaluate a device's status (`_STA`) to decide if it is present. An absent status
|
||||||
/// (`_STA`) means present by default; an evaluation failure is treated as present too (we'd
|
/// (`_STA`) means present by default; an evaluation failure is treated as present too (we'd
|
||||||
/// rather over-report than hide a device we couldn't introspect).
|
/// rather over-report than hide a device we couldn't introspect).
|
||||||
fn devicePresent(ev: *aml.Interp, node: *aml.Node) bool {
|
fn devicePresent(interpreter: *aml.Interpreter, node: *aml.Node) bool {
|
||||||
const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true;
|
const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true;
|
||||||
const obj = ev.evaluate(sta, &.{}) catch return true;
|
const obj = interpreter.evaluate(sta, &.{}) catch return true;
|
||||||
const status = obj.asInt() catch return true;
|
const status = obj.asInteger() catch return true;
|
||||||
return (status & 0x01) != 0; // bit 0 = present
|
return (status & 0x01) != 0; // bit 0 = present
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The first PCI host bridge in the generic tree (segment 0).
|
/// The first PCI host bridge in the generic tree (segment 0).
|
||||||
fn matchHostBridge(dt: *DeviceTree) ?*device.Device {
|
fn matchHostBridge(device_tree: *DeviceTree) ?*device_model.Device {
|
||||||
var c = dt.root.first_child;
|
var c = device_tree.root.first_child;
|
||||||
while (c) |ch| : (c = ch.next_sibling) {
|
while (c) |ch| : (c = ch.next_sibling) {
|
||||||
if (ch.class == .pci_host_bridge) return ch;
|
if (ch.class == .pci_host_bridge) return ch;
|
||||||
}
|
}
|
||||||
@@ -788,12 +826,12 @@ fn matchHostBridge(dt: *DeviceTree) ?*device.Device {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// The PCI function node under `bridge` at the address the device's address object
|
/// The PCI function node under `bridge` at the address the device's address object
|
||||||
/// (`_ADR`) names (dev/func on
|
/// (`_ADR`) names (device/function on
|
||||||
/// `bus`), or null.
|
/// `bus`), or null.
|
||||||
fn findPciNode(bridge: *device.Device, bus: u8, adr: u32) ?*device.Device {
|
fn findPciNode(bridge: *device_model.Device, bus: u8, adr: u32) ?*device_model.Device {
|
||||||
const dev: u16 = @truncate((adr >> 16) & 0x1F);
|
const device: u16 = @truncate((adr >> 16) & 0x1F);
|
||||||
const func: u16 = @truncate(adr & 0x7);
|
const function: u16 = @truncate(adr & 0x7);
|
||||||
const target: u16 = (@as(u16, bus) << 8) | (dev << 3) | func;
|
const target: u16 = (@as(u16, bus) << 8) | (device << 3) | function;
|
||||||
var c = bridge.first_child;
|
var c = bridge.first_child;
|
||||||
while (c) |ch| : (c = ch.next_sibling) {
|
while (c) |ch| : (c = ch.next_sibling) {
|
||||||
if (ch.ids.pci_bdf) |bdf| {
|
if (ch.ids.pci_bdf) |bdf| {
|
||||||
@@ -832,13 +870,13 @@ fn isPciRootNode(node: *aml.Node) bool {
|
|||||||
/// Read a device's hardware ID (`_HID`) into the generic device: an integer decodes as an EISA
|
/// Read a device's hardware ID (`_HID`) into the generic device: an integer decodes as an EISA
|
||||||
/// id ("PNP0A03"), a string is taken verbatim. Handles both the common static
|
/// id ("PNP0A03"), a string is taken verbatim. Handles both the common static
|
||||||
/// Name form and a Method form (evaluated).
|
/// Name form and a Method form (evaluated).
|
||||||
fn applyHid(dev: *device.Device, node: *aml.Node, ev: *aml.Interp) void {
|
fn applyHid(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||||
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return;
|
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return;
|
||||||
if (hid.kind == .method) {
|
if (hid.kind == .method) {
|
||||||
const obj = ev.evaluate(hid, &.{}) catch return;
|
const obj = interpreter.evaluate(hid, &.{}) catch return;
|
||||||
switch (obj) {
|
switch (obj) {
|
||||||
.integer => |n| setEisaHid(dev, @truncate(n)),
|
.integer => |n| setEisaHid(device, @truncate(n)),
|
||||||
.string => |s| dev.setHid(s),
|
.string => |s| device.setHid(s),
|
||||||
else => {},
|
else => {},
|
||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
@@ -849,34 +887,34 @@ fn applyHid(dev: *device.Device, node: *aml.Node, ev: *aml.Interp) void {
|
|||||||
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||||
var p: usize = 0;
|
var p: usize = 0;
|
||||||
const n = readIntObj(v, &p) orelse return;
|
const n = readIntObj(v, &p) orelse return;
|
||||||
setEisaHid(dev, @truncate(n));
|
setEisaHid(device, @truncate(n));
|
||||||
},
|
},
|
||||||
0x0D => dev.setHid(cstr(v[1..])), // StringPrefix
|
0x0D => device.setHid(cstr(v[1..])), // StringPrefix
|
||||||
else => {},
|
else => {},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn setEisaHid(dev: *device.Device, id: u32) void {
|
fn setEisaHid(device: *device_model.Device, id: u32) void {
|
||||||
dev.ids.acpi_hid = id;
|
device.ids.acpi_hid = id;
|
||||||
var buf: [8]u8 = undefined;
|
var buffer: [8]u8 = undefined;
|
||||||
dev.setHid(eisaIdToStr(id, &buf));
|
device.setHid(eisaIdToStr(id, &buffer));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Parse a device's current resource settings (`_CRS`). The evaluator handles both the static
|
/// Parse a device's current resource settings (`_CRS`). The evaluator handles both the static
|
||||||
/// `Buffer` form (a `Name`) and the method form uniformly, yielding the
|
/// `Buffer` form (a `Name`) and the method form uniformly, yielding the
|
||||||
/// ResourceTemplate bytes we then decode.
|
/// ResourceTemplate bytes we then decode.
|
||||||
fn applyCrs(dev: *device.Device, node: *aml.Node, ev: *aml.Interp) void {
|
fn applyCrs(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||||
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
||||||
const obj = ev.evaluate(crs, &.{}) catch return;
|
const obj = interpreter.evaluate(crs, &.{}) catch return;
|
||||||
const buf = switch (obj) {
|
const buffer = switch (obj) {
|
||||||
.buffer => |b| b,
|
.buffer => |b| b,
|
||||||
else => return,
|
else => return,
|
||||||
};
|
};
|
||||||
parseResourceTemplate(dev, buf);
|
parseResourceTemplate(device, buffer);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Walk a ResourceTemplate byte list, adding recognised descriptors as resources.
|
/// Walk a ResourceTemplate byte list, adding recognised descriptors as resources.
|
||||||
fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void {
|
fn parseResourceTemplate(device: *device_model.Device, bytes: []const u8) void {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < bytes.len) {
|
while (i < bytes.len) {
|
||||||
const tag = bytes[i];
|
const tag = bytes[i];
|
||||||
@@ -890,14 +928,14 @@ fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void {
|
|||||||
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
||||||
var b: usize = 0;
|
var b: usize = 0;
|
||||||
while (b < 16) : (b += 1) {
|
while (b < 16) : (b += 1) {
|
||||||
if (mask & (@as(u16, 1) << @intCast(b)) != 0) _ = dev.addResource(.irq, b, 1);
|
if (mask & (@as(u16, 1) << @intCast(b)) != 0) _ = device.addResource(.irq, b, 1);
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
0x08 => if (len >= 7) { // IO port: min at +1, length at +6
|
0x08 => if (len >= 7) { // IO port: minimum at +1, length at +6
|
||||||
_ = dev.addResource(.io_port, rd16(bytes, body + 1), bytes[body + 6]);
|
_ = device.addResource(.io_port, rd16(bytes, body + 1), bytes[body + 6]);
|
||||||
},
|
},
|
||||||
0x09 => if (len >= 3) { // Fixed IO: base at +0, length at +2
|
0x09 => if (len >= 3) { // Fixed IO: base at +0, length at +2
|
||||||
_ = dev.addResource(.io_port, rd16(bytes, body), bytes[body + 2]);
|
_ = device.addResource(.io_port, rd16(bytes, body), bytes[body + 2]);
|
||||||
},
|
},
|
||||||
0x0F => break, // EndTag
|
0x0F => break, // EndTag
|
||||||
else => {},
|
else => {},
|
||||||
@@ -910,20 +948,20 @@ fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void {
|
|||||||
const body = i + 3;
|
const body = i + 3;
|
||||||
if (body + len > bytes.len) break;
|
if (body + len > bytes.len) break;
|
||||||
switch (tag) {
|
switch (tag) {
|
||||||
0x85 => if (len >= 17) { // Memory32: min at +1, length at +13
|
0x85 => if (len >= 17) { // Memory32: minimum at +1, length at +13
|
||||||
_ = dev.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 13));
|
_ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 13));
|
||||||
},
|
},
|
||||||
0x86 => if (len >= 9) { // Memory32Fixed: base at +1, length at +5
|
0x86 => if (len >= 9) { // Memory32Fixed: base at +1, length at +5
|
||||||
_ = dev.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 5));
|
_ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 5));
|
||||||
},
|
},
|
||||||
0x89 => if (len >= 2) { // Extended IRQ: count at +1, then count u32s
|
0x89 => if (len >= 2) { // Extended IRQ: count at +1, then count u32s
|
||||||
const count = bytes[body + 1];
|
const count = bytes[body + 1];
|
||||||
var k: usize = 0;
|
var k: usize = 0;
|
||||||
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
||||||
_ = dev.addResource(.irq, rd32(bytes, body + 2 + k * 4), 1);
|
_ = device.addResource(.irq, rd32(bytes, body + 2 + k * 4), 1);
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
0x87, 0x88, 0x8A => parseAddressSpace(dev, tag, bytes[body .. body + len]),
|
0x87, 0x88, 0x8A => parseAddressSpace(device, tag, bytes[body .. body + len]),
|
||||||
else => {},
|
else => {},
|
||||||
}
|
}
|
||||||
i = body + len;
|
i = body + len;
|
||||||
@@ -932,39 +970,39 @@ fn parseResourceTemplate(dev: *device.Device, bytes: []const u8) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Word/DWord/QWord address-space descriptors: resource type at [0], then
|
/// Word/DWord/QWord address-space descriptors: resource type at [0], then
|
||||||
/// granularity/min/max/translation/length, each of width `w`.
|
/// granularity/minimum/maximum/translation/length, each of width `w`.
|
||||||
fn parseAddressSpace(dev: *device.Device, tag: u8, body: []const u8) void {
|
fn parseAddressSpace(device: *device_model.Device, tag: u8, body: []const u8) void {
|
||||||
const w: usize = switch (tag) {
|
const w: usize = switch (tag) {
|
||||||
0x88 => 2, // Word
|
0x88 => 2, // Word
|
||||||
0x87 => 4, // DWord
|
0x87 => 4, // DWord
|
||||||
else => 8, // QWord (0x8A)
|
else => 8, // QWord (0x8A)
|
||||||
};
|
};
|
||||||
if (body.len < 3 + 5 * w) return;
|
if (body.len < 3 + 5 * w) return;
|
||||||
const min = readN(body, 3 + w, w);
|
const minimum = readN(body, 3 + w, w);
|
||||||
const length = readN(body, 3 + 4 * w, w);
|
const length = readN(body, 3 + 4 * w, w);
|
||||||
const kind: device.ResourceKind = switch (body[0]) {
|
const kind: device_model.ResourceKind = switch (body[0]) {
|
||||||
0 => .memory,
|
0 => .memory,
|
||||||
1 => .io_port,
|
1 => .io_port,
|
||||||
else => .bus_range,
|
else => .bus_range,
|
||||||
};
|
};
|
||||||
_ = dev.addResource(kind, min, length);
|
_ = device.addResource(kind, minimum, length);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decode a packed EISA id into its 7-char string (e.g. 0x030AD041 -> "PNP0A03").
|
/// Decode a packed EISA id into its 7-char string (e.g. 0x030AD041 -> "PNP0A03").
|
||||||
fn eisaIdToStr(id: u32, buf: *[8]u8) []const u8 {
|
fn eisaIdToStr(id: u32, buffer: *[8]u8) []const u8 {
|
||||||
const b0: u16 = @intCast(id & 0xFF);
|
const b0: u16 = @intCast(id & 0xFF);
|
||||||
const b1: u16 = @intCast((id >> 8) & 0xFF);
|
const b1: u16 = @intCast((id >> 8) & 0xFF);
|
||||||
const b2: u8 = @truncate(id >> 16);
|
const b2: u8 = @truncate(id >> 16);
|
||||||
const b3: u8 = @truncate(id >> 24);
|
const b3: u8 = @truncate(id >> 24);
|
||||||
const mfg = (b0 << 8) | b1;
|
const mfg = (b0 << 8) | b1;
|
||||||
buf[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F));
|
buffer[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F));
|
||||||
buf[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F));
|
buffer[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F));
|
||||||
buf[2] = '@' + @as(u8, @intCast(mfg & 0x1F));
|
buffer[2] = '@' + @as(u8, @intCast(mfg & 0x1F));
|
||||||
buf[3] = hexDigit((b2 >> 4) & 0xF);
|
buffer[3] = hexDigit((b2 >> 4) & 0xF);
|
||||||
buf[4] = hexDigit(b2 & 0xF);
|
buffer[4] = hexDigit(b2 & 0xF);
|
||||||
buf[5] = hexDigit((b3 >> 4) & 0xF);
|
buffer[5] = hexDigit((b3 >> 4) & 0xF);
|
||||||
buf[6] = hexDigit(b3 & 0xF);
|
buffer[6] = hexDigit(b3 & 0xF);
|
||||||
return buf[0..7];
|
return buffer[0..7];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn hexDigit(n: u8) u8 {
|
fn hexDigit(n: u8) u8 {
|
||||||
@@ -976,13 +1014,13 @@ fn seg4(comptime s: *const [4:0]u8) [4]u8 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn cstr(bytes: []const u8) []const u8 {
|
fn cstr(bytes: []const u8) []const u8 {
|
||||||
const idx = std.mem.indexOfScalar(u8, bytes, 0) orelse bytes.len;
|
const index = std.mem.indexOfScalar(u8, bytes, 0) orelse bytes.len;
|
||||||
return bytes[0..idx];
|
return bytes[0..index];
|
||||||
}
|
}
|
||||||
|
|
||||||
const PkgLen = struct { value: usize, size: usize };
|
const PkgLen = struct { value: usize, size: usize };
|
||||||
|
|
||||||
fn pkgLen(bytes: []const u8, p: usize) ?PkgLen {
|
fn packageLength(bytes: []const u8, p: usize) ?PkgLen {
|
||||||
if (p >= bytes.len) return null;
|
if (p >= bytes.len) return null;
|
||||||
const lead = bytes[p];
|
const lead = bytes[p];
|
||||||
const follow: usize = lead >> 6;
|
const follow: usize = lead >> 6;
|
||||||
@@ -1049,9 +1087,9 @@ fn fadt(comptime T: type, base: [*]align(1) const u8, len: usize, off: usize) ?T
|
|||||||
return rd(T, base, off);
|
return rd(T, base, off);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decode a Generic Address Structure at `off` into a `RegAccess`. GAS layout:
|
/// Decode a Generic Address Structure at `off` into a `RegisterAccess`. GAS layout:
|
||||||
/// address_space(u8), bit_width(u8), bit_offset(u8), access_size(u8), address(u64).
|
/// address_space(u8), bit_width(u8), bit_offset(u8), access_size(u8), address(u64).
|
||||||
fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegAccess {
|
fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegisterAccess {
|
||||||
if (off + 12 > len) return null;
|
if (off + 12 > len) return null;
|
||||||
const address_space = rd(u8, base, off);
|
const address_space = rd(u8, base, off);
|
||||||
const bit_width = rd(u8, base, off + 1);
|
const bit_width = rd(u8, base, off + 1);
|
||||||
@@ -1065,7 +1103,7 @@ fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegAccess {
|
|||||||
|
|
||||||
/// A PM1 control register: prefer the 64-bit-capable X_ GAS form; fall back to the
|
/// A PM1 control register: prefer the 64-bit-capable X_ GAS form; fall back to the
|
||||||
/// legacy 32-bit I/O-port field. Width comes from PM1_CNT_LEN either way.
|
/// legacy 32-bit I/O-port field. Width comes from PM1_CNT_LEN either way.
|
||||||
fn readCntReg(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegAccess {
|
fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegisterAccess {
|
||||||
if (readGas(base, len, xoff)) |g| {
|
if (readGas(base, len, xoff)) |g| {
|
||||||
if (g.address != 0) return .{ .mmio = g.mmio, .address = g.address, .width = width };
|
if (g.address != 0) return .{ .mmio = g.mmio, .address = g.address, .width = width };
|
||||||
}
|
}
|
||||||
@@ -1073,16 +1111,16 @@ fn readCntReg(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: u
|
|||||||
return .{ .mmio = false, .address = port, .width = width };
|
return .{ .mmio = false, .address = port, .width = width };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The mapped config space of one PCI function (its 4 KiB ECAM page). Mapped
|
/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped
|
||||||
/// writable so BAR sizing can probe it; reads and writes both go through here.
|
/// writable so BAR sizing can probe it; reads and writes both go through here.
|
||||||
fn pciConfigPtr(alloc: McfgAllocation, hal: Hal, bus: u8, dev: u8, func: u8) [*]align(1) u8 {
|
fn pciConfigurationPtr(alloc: McfgAllocation, hal: Hal, bus: u8, device: u8, function: u8) [*]align(1) u8 {
|
||||||
const phys = alloc.base_address +
|
const physical = alloc.base_address +
|
||||||
(@as(u64, bus - alloc.start_bus) << 20) +
|
(@as(u64, bus - alloc.start_bus) << 20) +
|
||||||
(@as(u64, dev) << 15) +
|
(@as(u64, device) << 15) +
|
||||||
(@as(u64, func) << 12);
|
(@as(u64, function) << 12);
|
||||||
// Map the config page (writable, for BAR sizing) and use the virtual
|
// Map the configuration page (writable, for BAR sizing) and use the virtual
|
||||||
// address the HAL hands back.
|
// address the HAL hands back.
|
||||||
return @ptrFromInt(hal.mapMmio(phys, danos.page_size, true));
|
return @ptrFromInt(hal.mapMmio(physical, danos.page_size, true));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||||
@@ -1101,30 +1139,30 @@ fn wr(comptime T: type, bytes: [*]align(1) u8, off: usize, value: T) void {
|
|||||||
// --- tests ------------------------------------------------------------------
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
test "eisaIdToStr decodes a packed EISA id" {
|
test "eisaIdToStr decodes a packed EISA id" {
|
||||||
var buf: [8]u8 = undefined;
|
var buffer: [8]u8 = undefined;
|
||||||
// 0x030AD041 is the well-known encoding of "PNP0A03" (PCI root bridge).
|
// 0x030AD041 is the well-known encoding of "PNP0A03" (PCI root bridge).
|
||||||
try std.testing.expectEqualStrings("PNP0A03", eisaIdToStr(0x030AD041, &buf));
|
try std.testing.expectEqualStrings("PNP0A03", eisaIdToStr(0x030AD041, &buffer));
|
||||||
}
|
}
|
||||||
|
|
||||||
test "parseResourceTemplate extracts IO, IRQ, and fixed memory" {
|
test "parseResourceTemplate extracts IO, IRQ, and fixed memory" {
|
||||||
// ResourceTemplate { IO(min 0x60, len 8), IRQ(4), Memory32Fixed(0xFED00000, 0x1000) }
|
// ResourceTemplate { IO(minimum 0x60, len 8), IRQ(4), Memory32Fixed(0xFED00000, 0x1000) }
|
||||||
const rt = [_]u8{
|
const runtime = [_]u8{
|
||||||
0x47, 0x01, 0x60, 0x00, 0x60, 0x00, 0x01, 0x08, // small IO descriptor
|
0x47, 0x01, 0x60, 0x00, 0x60, 0x00, 0x01, 0x08, // small IO descriptor
|
||||||
0x22, 0x10, 0x00, // small IRQ descriptor (mask bit 4 -> IRQ 4)
|
0x22, 0x10, 0x00, // small IRQ descriptor (mask bit 4 -> IRQ 4)
|
||||||
0x86, 0x09, 0x00, 0x01, 0x00, 0x00, 0xD0, 0xFE, 0x00, 0x10, 0x00, 0x00, // Memory32Fixed
|
0x86, 0x09, 0x00, 0x01, 0x00, 0x00, 0xD0, 0xFE, 0x00, 0x10, 0x00, 0x00, // Memory32Fixed
|
||||||
0x79, 0x00, // EndTag
|
0x79, 0x00, // EndTag
|
||||||
};
|
};
|
||||||
var dev = device.Device{};
|
var device = device_model.Device{};
|
||||||
parseResourceTemplate(&dev, &rt);
|
parseResourceTemplate(&device, &runtime);
|
||||||
|
|
||||||
try std.testing.expectEqual(@as(u8, 3), dev.resource_count);
|
try std.testing.expectEqual(@as(u8, 3), device.resource_count);
|
||||||
const rs = dev.resources[0..dev.resource_count];
|
const rs = device.resources[0..device.resource_count];
|
||||||
try std.testing.expectEqual(device.ResourceKind.io_port, rs[0].kind);
|
try std.testing.expectEqual(device_model.ResourceKind.io_port, rs[0].kind);
|
||||||
try std.testing.expectEqual(@as(u64, 0x60), rs[0].start);
|
try std.testing.expectEqual(@as(u64, 0x60), rs[0].start);
|
||||||
try std.testing.expectEqual(@as(u64, 8), rs[0].len);
|
try std.testing.expectEqual(@as(u64, 8), rs[0].len);
|
||||||
try std.testing.expectEqual(device.ResourceKind.irq, rs[1].kind);
|
try std.testing.expectEqual(device_model.ResourceKind.irq, rs[1].kind);
|
||||||
try std.testing.expectEqual(@as(u64, 4), rs[1].start);
|
try std.testing.expectEqual(@as(u64, 4), rs[1].start);
|
||||||
try std.testing.expectEqual(device.ResourceKind.memory, rs[2].kind);
|
try std.testing.expectEqual(device_model.ResourceKind.memory, rs[2].kind);
|
||||||
try std.testing.expectEqual(@as(u64, 0xFED00000), rs[2].start);
|
try std.testing.expectEqual(@as(u64, 0xFED00000), rs[2].start);
|
||||||
try std.testing.expectEqual(@as(u64, 0x1000), rs[2].len);
|
try std.testing.expectEqual(@as(u64, 0x1000), rs[2].len);
|
||||||
}
|
}
|
||||||
|
|||||||
+50
-52
@@ -9,20 +9,18 @@
|
|||||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const op = @import("opcodes.zig");
|
const opcode = @import("opcodes.zig");
|
||||||
const parser = @import("parser.zig");
|
const parser = @import("parser.zig");
|
||||||
const namespace = @import("namespace.zig");
|
|
||||||
const interp = @import("interp.zig");
|
|
||||||
|
|
||||||
pub const Namespace = namespace.Namespace;
|
pub const Namespace = @import("namespace.zig").Namespace;
|
||||||
pub const Node = namespace.Node;
|
pub const Node = @import("namespace.zig").Node;
|
||||||
pub const NodeKind = namespace.NodeKind;
|
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||||
/// enough for device discovery. See `interp.zig`.
|
/// enough for device discovery. See `interp.zig`.
|
||||||
pub const Interp = interp.Interp;
|
pub const Interpreter = @import("interp.zig").Interpreter;
|
||||||
pub const Object = interp.Object;
|
pub const Object = @import("interp.zig").Object;
|
||||||
pub const EvalHal = interp.Hal;
|
pub const EvaluateHal = @import("interp.zig").Hal;
|
||||||
|
|
||||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||||
pub const SleepType = struct {
|
pub const SleepType = struct {
|
||||||
@@ -41,22 +39,22 @@ pub const ParseResult = struct {
|
|||||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||||
var ns = try Namespace.init(allocator);
|
var namespace = try Namespace.init(allocator);
|
||||||
var consumed: usize = 0;
|
var consumed: usize = 0;
|
||||||
var total: usize = 0;
|
var total: usize = 0;
|
||||||
for (blocks) |block| {
|
for (blocks) |block| {
|
||||||
var p = parser.Parser.init(block, &ns);
|
var p = parser.Parser.init(block, &namespace);
|
||||||
consumed += p.parseAll();
|
consumed += p.parseAll();
|
||||||
total += block.len;
|
total += block.len;
|
||||||
}
|
}
|
||||||
return .{ .namespace = ns, .consumed = consumed, .total = total };
|
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||||
pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||||
const seg = [4]u8{ '_', 'S', '0' + state, '_' };
|
const segment = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||||
const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null;
|
const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null;
|
||||||
if (node.kind != .name) return null;
|
if (node.kind != .name) return null;
|
||||||
return parseSleepPackage(node.value);
|
return parseSleepPackage(node.value);
|
||||||
}
|
}
|
||||||
@@ -64,20 +62,20 @@ pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
|||||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||||
if (value.len == 0 or value[0] != op.package_op) return null;
|
if (value.len == 0 or value[0] != opcode.package_opcode) return null;
|
||||||
var p: usize = 1;
|
var p: usize = 1;
|
||||||
p += pkgLengthSize(value, p) orelse return null;
|
p += packageLengthSize(value, p) orelse return null;
|
||||||
if (p >= value.len) return null;
|
if (p >= value.len) return null;
|
||||||
const num_elements = value[p];
|
const number_elements = value[p];
|
||||||
p += 1;
|
p += 1;
|
||||||
|
|
||||||
const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||||
fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
fn packageLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||||
if (p >= bytes.len) return null;
|
if (p >= bytes.len) return null;
|
||||||
const follow: usize = bytes[p] >> 6;
|
const follow: usize = bytes[p] >> 6;
|
||||||
if (p + 1 + follow > bytes.len) return null;
|
if (p + 1 + follow > bytes.len) return null;
|
||||||
@@ -87,16 +85,16 @@ fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
|||||||
/// Read one AML integer data object at `p`, advancing `p`.
|
/// Read one AML integer data object at `p`, advancing `p`.
|
||||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||||
if (p.* >= bytes.len) return null;
|
if (p.* >= bytes.len) return null;
|
||||||
const opcode = bytes[p.*];
|
const opcode_byte = bytes[p.*];
|
||||||
p.* += 1;
|
p.* += 1;
|
||||||
return switch (opcode) {
|
return switch (opcode_byte) {
|
||||||
op.zero_op => 0,
|
opcode.zero_opcode => 0,
|
||||||
op.one_op => 1,
|
opcode.one_opcode => 1,
|
||||||
op.ones_op => 0xFF,
|
opcode.ones_opcode => 0xFF,
|
||||||
op.byte_prefix => readLittle(bytes, p, 1),
|
opcode.byte_prefix => readLittle(bytes, p, 1),
|
||||||
op.word_prefix => readLittle(bytes, p, 2),
|
opcode.word_prefix => readLittle(bytes, p, 2),
|
||||||
op.dword_prefix => readLittle(bytes, p, 4),
|
opcode.dword_prefix => readLittle(bytes, p, 4),
|
||||||
op.qword_prefix => readLittle(bytes, p, 8),
|
opcode.qword_prefix => readLittle(bytes, p, 8),
|
||||||
else => null,
|
else => null,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -125,19 +123,19 @@ test "parses a nested namespace and finds the sleep package" {
|
|||||||
const blob = [_]u8{
|
const blob = [_]u8{
|
||||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||||
// Scope(\_SB) pkglen=0x27
|
// Scope(\_SB) packagelen=0x27
|
||||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||||
// Device(PCI0) pkglen=0x1F
|
// Device(PCI0) packagelen=0x1F
|
||||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
||||||
// Name(_HID, 0x11)
|
// Name(_HID, 0x11)
|
||||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||||
// Method(MTHD, flags=1) empty, pkglen=0x06
|
// Method(MTHD, flags=1) empty, packagelen=0x06
|
||||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
||||||
// Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B
|
// Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B
|
||||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
||||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
||||||
// Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B
|
// Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B
|
||||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -149,32 +147,32 @@ test "parses a nested namespace and finds the sleep package" {
|
|||||||
try std.testing.expectEqual(blob.len, result.consumed);
|
try std.testing.expectEqual(blob.len, result.consumed);
|
||||||
try std.testing.expectEqual(blob.len, result.total);
|
try std.testing.expectEqual(blob.len, result.total);
|
||||||
|
|
||||||
const ns = &result.namespace;
|
const namespace = &result.namespace;
|
||||||
|
|
||||||
// Expected top-level nodes.
|
// Expected top-level nodes.
|
||||||
const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||||
const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||||
_ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
_ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||||
|
|
||||||
// The 1-arg method's arg count was parsed from its flags byte.
|
// The 1-arg method's arg count was parsed from its flags byte.
|
||||||
const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||||
|
|
||||||
// OperationRegion and the Field unit made it into the namespace.
|
// OperationRegion and the Field unit made it into the namespace.
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||||
|
|
||||||
// The sleep package decoded.
|
// The sleep package decoded.
|
||||||
const s5 = sleepState(ns, 5) orelse return error.NoS5;
|
const s5 = sleepState(namespace, 5) orelse return error.NoS5;
|
||||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn noMap(phys: u64, _: u64, _: bool) u64 {
|
fn noMap(physical: u64, _: u64, _: bool) u64 {
|
||||||
return phys;
|
return physical;
|
||||||
}
|
}
|
||||||
fn noRead(_: u8, _: u16) u32 {
|
fn noRead(_: u8, _: u16) u32 {
|
||||||
return 0;
|
return 0;
|
||||||
@@ -198,13 +196,13 @@ test "interpreter runs a method with args, arithmetic, and control flow" {
|
|||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
defer arena.deinit();
|
defer arena.deinit();
|
||||||
var result = try parse(arena.allocator(), &.{&blob});
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
const ns = &result.namespace;
|
const namespace = &result.namespace;
|
||||||
const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||||
|
|
||||||
var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||||
|
|
||||||
const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInt());
|
try std.testing.expectEqual(@as(u64, 1), try hi.asInteger());
|
||||||
const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInt());
|
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||||
}
|
}
|
||||||
|
|||||||
+298
-299
@@ -14,14 +14,13 @@
|
|||||||
//! fall back — never a hard failure.
|
//! fall back — never a hard failure.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const op = @import("opcodes.zig");
|
const opcode = @import("opcodes.zig");
|
||||||
const nsp = @import("namespace.zig");
|
const Node = @import("namespace.zig").Node;
|
||||||
const Node = nsp.Node;
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
const Namespace = nsp.Namespace;
|
|
||||||
|
|
||||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio).
|
||||||
pub const Hal = struct {
|
pub const Hal = struct {
|
||||||
mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64,
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
};
|
};
|
||||||
@@ -37,7 +36,7 @@ pub const Object = union(enum) {
|
|||||||
package: []Object,
|
package: []Object,
|
||||||
reference: *Node,
|
reference: *Node,
|
||||||
|
|
||||||
pub fn asInt(self: Object) Error!u64 {
|
pub fn asInteger(self: Object) Error!u64 {
|
||||||
return switch (self) {
|
return switch (self) {
|
||||||
.integer => |v| v,
|
.integer => |v| v,
|
||||||
.buffer => |b| blk: {
|
.buffer => |b| blk: {
|
||||||
@@ -53,14 +52,14 @@ pub const Object = union(enum) {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const max_segs = 16;
|
const maximum_segments = 16;
|
||||||
const NamePath = struct {
|
const NamePath = struct {
|
||||||
rooted: bool = false,
|
rooted: bool = false,
|
||||||
parents: u8 = 0,
|
parents: u8 = 0,
|
||||||
segs: [max_segs][4]u8 = undefined,
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
count: usize = 0,
|
count: usize = 0,
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
return self.segs[0..self.count];
|
return self.segments[0..self.count];
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -86,7 +85,7 @@ const Cursor = struct {
|
|||||||
self.i += n;
|
self.i += n;
|
||||||
return s;
|
return s;
|
||||||
}
|
}
|
||||||
fn pkgLen(self: *Cursor) Error!usize {
|
fn packageLength(self: *Cursor) Error!usize {
|
||||||
const lead = try self.byte();
|
const lead = try self.byte();
|
||||||
const follow: usize = lead >> 6;
|
const follow: usize = lead >> 6;
|
||||||
if (follow == 0) return lead & 0x3F;
|
if (follow == 0) return lead & 0x3F;
|
||||||
@@ -96,36 +95,36 @@ const Cursor = struct {
|
|||||||
return value;
|
return value;
|
||||||
}
|
}
|
||||||
fn nameString(self: *Cursor) Error!NamePath {
|
fn nameString(self: *Cursor) Error!NamePath {
|
||||||
var np = NamePath{};
|
var name_path = NamePath{};
|
||||||
if (self.peek() == op.root_char) {
|
if (self.peek() == opcode.root_char) {
|
||||||
np.rooted = true;
|
name_path.rooted = true;
|
||||||
self.i += 1;
|
self.i += 1;
|
||||||
} else {
|
} else {
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1;
|
while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1;
|
||||||
}
|
}
|
||||||
const lead = self.peek() orelse return np;
|
const lead = self.peek() orelse return name_path;
|
||||||
switch (lead) {
|
switch (lead) {
|
||||||
0x00 => self.i += 1,
|
0x00 => self.i += 1,
|
||||||
op.dual_name_prefix => {
|
opcode.dual_name_prefix => {
|
||||||
self.i += 1;
|
self.i += 1;
|
||||||
try self.seg(&np);
|
try self.segment(&name_path);
|
||||||
try self.seg(&np);
|
try self.segment(&name_path);
|
||||||
},
|
},
|
||||||
op.multi_name_prefix => {
|
opcode.multi_name_prefix => {
|
||||||
self.i += 1;
|
self.i += 1;
|
||||||
const cnt = try self.byte();
|
const count = try self.byte();
|
||||||
var k: usize = 0;
|
var k: usize = 0;
|
||||||
while (k < cnt) : (k += 1) try self.seg(&np);
|
while (k < count) : (k += 1) try self.segment(&name_path);
|
||||||
},
|
},
|
||||||
else => try self.seg(&np),
|
else => try self.segment(&name_path),
|
||||||
}
|
}
|
||||||
return np;
|
return name_path;
|
||||||
}
|
}
|
||||||
fn seg(self: *Cursor, np: *NamePath) Error!void {
|
fn segment(self: *Cursor, name_path: *NamePath) Error!void {
|
||||||
const s = try self.take(4);
|
const s = try self.take(4);
|
||||||
if (np.count < max_segs) {
|
if (name_path.count < maximum_segments) {
|
||||||
np.segs[np.count] = s[0..4].*;
|
name_path.segments[name_path.count] = s[0..4].*;
|
||||||
np.count += 1;
|
name_path.count += 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -140,45 +139,45 @@ const Frame = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// A CreateField binding: a name that indexes into a buffer object.
|
/// A CreateField binding: a name that indexes into a buffer object.
|
||||||
const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 };
|
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||||
|
|
||||||
pub const Interp = struct {
|
pub const Interpreter = struct {
|
||||||
ns: *Namespace,
|
namespace: *Namespace,
|
||||||
hal: Hal,
|
hal: Hal,
|
||||||
arena: std.mem.Allocator,
|
arena: std.mem.Allocator,
|
||||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||||
dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||||
/// CreateField bindings active for the current evaluation.
|
/// CreateField bindings active for the current evaluation.
|
||||||
fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{},
|
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||||
|
|
||||||
pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp {
|
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||||
return .{ .ns = ns, .hal = hal, .arena = arena };
|
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||||
/// Field. Resets per-evaluation runtime state first.
|
/// Field. Resets per-evaluation runtime state first.
|
||||||
pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
self.dyn.clearRetainingCapacity();
|
self.dynamic_overrides.clearRetainingCapacity();
|
||||||
self.fields.clearRetainingCapacity();
|
self.fields.clearRetainingCapacity();
|
||||||
return self.invoke(node, args);
|
return self.invoke(node, args);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
switch (node.kind) {
|
switch (node.kind) {
|
||||||
.method => {
|
.method => {
|
||||||
var frame = Frame{ .scope = node };
|
var frame = Frame{ .scope = node };
|
||||||
for (args, 0..) |a, i| {
|
for (args, 0..) |a, i| {
|
||||||
if (i < frame.args.len) frame.args[i] = a;
|
if (i < frame.args.len) frame.args[i] = a;
|
||||||
}
|
}
|
||||||
var cur = Cursor{ .b = node.value };
|
var current = Cursor{ .b = node.value };
|
||||||
try self.execList(&cur, &frame);
|
try self.executeList(¤t, &frame);
|
||||||
return frame.ret;
|
return frame.ret;
|
||||||
},
|
},
|
||||||
.name => {
|
.name => {
|
||||||
if (self.dyn.get(node)) |o| return o;
|
if (self.dynamic_overrides.get(node)) |o| return o;
|
||||||
var cur = Cursor{ .b = node.value };
|
var current = Cursor{ .b = node.value };
|
||||||
var frame = Frame{ .scope = node.parent orelse self.ns.root };
|
var frame = Frame{ .scope = node.parent orelse self.namespace.root };
|
||||||
return self.term(&cur, &frame);
|
return self.term(¤t, &frame);
|
||||||
},
|
},
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
else => return .{ .reference = node },
|
else => return .{ .reference = node },
|
||||||
@@ -186,96 +185,96 @@ pub const Interp = struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||||
fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void {
|
fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void {
|
||||||
while (!cur.eof() and !frame.returned and !frame.broke) {
|
while (!current.eof() and !frame.returned and !frame.broke) {
|
||||||
_ = try self.term(cur, frame);
|
_ = try self.term(current, frame);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||||
/// statements).
|
/// statements).
|
||||||
fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
if (isNameStart(lead)) return self.nameRef(cur, frame);
|
if (isNameStart(lead)) return self.nameReference(current, frame);
|
||||||
_ = try cur.byte();
|
_ = try current.byte();
|
||||||
|
|
||||||
return switch (lead) {
|
return switch (lead) {
|
||||||
op.zero_op => Object{ .integer = 0 },
|
opcode.zero_opcode => Object{ .integer = 0 },
|
||||||
op.one_op => Object{ .integer = 1 },
|
opcode.one_opcode => Object{ .integer = 1 },
|
||||||
op.ones_op => Object{ .integer = ~@as(u64, 0) },
|
opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) },
|
||||||
op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) },
|
opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) },
|
||||||
op.word_prefix => Object{ .integer = try self.readConst(cur, 2) },
|
opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) },
|
||||||
op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) },
|
opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) },
|
||||||
op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) },
|
opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) },
|
||||||
op.string_prefix => try self.readString(cur),
|
opcode.string_prefix => try self.readString(current),
|
||||||
op.buffer_op => try self.buffer(cur, frame),
|
opcode.buffer_opcode => try self.buffer(current, frame),
|
||||||
op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op),
|
opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode),
|
||||||
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op],
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode],
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op],
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode],
|
||||||
|
|
||||||
op.return_op => blk: {
|
opcode.return_opcode => blk: {
|
||||||
frame.ret = try self.term(cur, frame);
|
frame.ret = try self.term(current, frame);
|
||||||
frame.returned = true;
|
frame.returned = true;
|
||||||
break :blk .uninitialized;
|
break :blk .uninitialized;
|
||||||
},
|
},
|
||||||
op.break_op => blk: {
|
opcode.break_opcode => blk: {
|
||||||
frame.broke = true;
|
frame.broke = true;
|
||||||
break :blk .uninitialized;
|
break :blk .uninitialized;
|
||||||
},
|
},
|
||||||
op.continue_op, op.noop_op => .uninitialized,
|
opcode.continue_opcode, opcode.noop_opcode => .uninitialized,
|
||||||
|
|
||||||
op.if_op => try self.ifElse(cur, frame),
|
opcode.if_opcode => try self.ifElse(current, frame),
|
||||||
op.while_op => try self.whileLoop(cur, frame),
|
opcode.while_opcode => try self.whileLoop(current, frame),
|
||||||
op.store_op => try self.store(cur, frame),
|
opcode.store_opcode => try self.store(current, frame),
|
||||||
op.increment_op => try self.incDec(cur, frame, 1),
|
opcode.increment_opcode => try self.incDec(current, frame, 1),
|
||||||
op.decrement_op => try self.incDec(cur, frame, -1),
|
opcode.decrement_opcode => try self.incDec(current, frame, -1),
|
||||||
|
|
||||||
op.add_op => try self.binary(cur, frame, .add),
|
opcode.add_opcode => try self.binary(current, frame, .add),
|
||||||
op.subtract_op => try self.binary(cur, frame, .sub),
|
opcode.subtract_opcode => try self.binary(current, frame, .sub),
|
||||||
op.multiply_op => try self.binary(cur, frame, .mul),
|
opcode.multiply_opcode => try self.binary(current, frame, .mul),
|
||||||
op.mod_op => try self.binary(cur, frame, .mod),
|
opcode.mod_opcode => try self.binary(current, frame, .mod),
|
||||||
op.and_op => try self.binary(cur, frame, .band),
|
opcode.and_opcode => try self.binary(current, frame, .band),
|
||||||
op.or_op => try self.binary(cur, frame, .bor),
|
opcode.or_opcode => try self.binary(current, frame, .bor),
|
||||||
op.xor_op => try self.binary(cur, frame, .bxor),
|
opcode.xor_opcode => try self.binary(current, frame, .bxor),
|
||||||
op.nand_op => try self.binary(cur, frame, .nand),
|
opcode.nand_opcode => try self.binary(current, frame, .nand),
|
||||||
op.nor_op => try self.binary(cur, frame, .nor),
|
opcode.nor_opcode => try self.binary(current, frame, .nor),
|
||||||
op.shift_left_op => try self.binary(cur, frame, .shl),
|
opcode.shift_left_opcode => try self.binary(current, frame, .shl),
|
||||||
op.shift_right_op => try self.binary(cur, frame, .shr),
|
opcode.shift_right_opcode => try self.binary(current, frame, .shr),
|
||||||
op.divide_op => try self.divide(cur, frame),
|
opcode.divide_opcode => try self.divide(current, frame),
|
||||||
|
|
||||||
op.land_op => try self.logic2(cur, frame, .land),
|
opcode.land_opcode => try self.logic2(current, frame, .land),
|
||||||
op.lor_op => try self.logic2(cur, frame, .lor),
|
opcode.lor_opcode => try self.logic2(current, frame, .lor),
|
||||||
op.lequal_op => try self.logic2(cur, frame, .eq),
|
opcode.lequal_opcode => try self.logic2(current, frame, .eq),
|
||||||
op.lgreater_op => try self.logic2(cur, frame, .gt),
|
opcode.lgreater_opcode => try self.logic2(current, frame, .gt),
|
||||||
op.lless_op => try self.logic2(cur, frame, .lt),
|
opcode.lless_opcode => try self.logic2(current, frame, .lt),
|
||||||
op.lnot_op => try self.lnot(cur, frame),
|
opcode.lnot_opcode => try self.lnot(current, frame),
|
||||||
|
|
||||||
op.not_op => blk: {
|
opcode.not_opcode => blk: {
|
||||||
const v = try self.evalInt(cur, frame);
|
const v = try self.evaluateInteger(current, frame);
|
||||||
const r = ~v;
|
const r = ~v;
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
break :blk .{ .integer = r };
|
break :blk .{ .integer = r };
|
||||||
},
|
},
|
||||||
|
|
||||||
op.size_of_op => try self.sizeOf(cur, frame),
|
opcode.size_of_opcode => try self.sizeOf(current, frame),
|
||||||
op.index_op => try self.index(cur, frame),
|
opcode.index_opcode => try self.index(current, frame),
|
||||||
op.deref_of_op => try self.derefOf(cur, frame),
|
opcode.dereference_of_opcode => try self.dereferenceOf(current, frame),
|
||||||
op.to_integer_op => blk: {
|
opcode.to_integer_opcode => blk: {
|
||||||
const v = try self.evalInt(cur, frame);
|
const v = try self.evaluateInteger(current, frame);
|
||||||
try self.storeTarget(cur, frame, .{ .integer = v });
|
try self.storeTarget(current, frame, .{ .integer = v });
|
||||||
break :blk .{ .integer = v };
|
break :blk .{ .integer = v };
|
||||||
},
|
},
|
||||||
op.to_buffer_op => try self.passThroughUnary(cur, frame),
|
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||||
|
|
||||||
op.ext_op_prefix => try self.ext(cur, frame),
|
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||||
|
|
||||||
// CreateXField: source, index, name (bit widths differ by op)
|
// CreateXField: source, index, name (bit widths differ by op)
|
||||||
op.create_bit_field_op => try self.createField(cur, frame, 1),
|
opcode.create_bit_field_opcode => try self.createField(current, frame, 1),
|
||||||
op.create_byte_field_op => try self.createField(cur, frame, 8),
|
opcode.create_byte_field_opcode => try self.createField(current, frame, 8),
|
||||||
op.create_word_field_op => try self.createField(cur, frame, 16),
|
opcode.create_word_field_opcode => try self.createField(current, frame, 16),
|
||||||
op.create_dword_field_op => try self.createField(cur, frame, 32),
|
opcode.create_dword_field_opcode => try self.createField(current, frame, 32),
|
||||||
op.create_qword_field_op => try self.createField(cur, frame, 64),
|
opcode.create_qword_field_opcode => try self.createField(current, frame, 64),
|
||||||
|
|
||||||
else => error.Unsupported,
|
else => error.Unsupported,
|
||||||
};
|
};
|
||||||
@@ -283,15 +282,15 @@ pub const Interp = struct {
|
|||||||
|
|
||||||
// --- name references ----------------------------------------------------
|
// --- name references ----------------------------------------------------
|
||||||
|
|
||||||
fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const np = try cur.nameString();
|
const name_path = try current.nameString();
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse
|
||||||
return .uninitialized; // unknown name -> treat as uninitialised
|
return .uninitialized; // unknown name -> treat as uninitialised
|
||||||
switch (node.kind) {
|
switch (node.kind) {
|
||||||
.method => {
|
.method => {
|
||||||
var argbuf: [7]Object = undefined;
|
var argbuf: [7]Object = undefined;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame);
|
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame);
|
||||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||||
},
|
},
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
@@ -302,58 +301,58 @@ pub const Interp = struct {
|
|||||||
|
|
||||||
// --- data objects -------------------------------------------------------
|
// --- data objects -------------------------------------------------------
|
||||||
|
|
||||||
fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 {
|
fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 {
|
||||||
_ = self;
|
_ = self;
|
||||||
const bytes = try cur.take(n);
|
const bytes = try current.take(n);
|
||||||
var v: u64 = 0;
|
var v: u64 = 0;
|
||||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||||
return v;
|
return v;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readString(self: *Interp, cur: *Cursor) Error!Object {
|
fn readString(self: *Interpreter, current: *Cursor) Error!Object {
|
||||||
const start = cur.i;
|
const start = current.i;
|
||||||
while (cur.peek()) |c| {
|
while (current.peek()) |c| {
|
||||||
cur.i += 1;
|
current.i += 1;
|
||||||
if (c == 0) break;
|
if (c == 0) break;
|
||||||
}
|
}
|
||||||
const raw = cur.b[start .. cur.i - 1];
|
const raw = current.b[start .. current.i - 1];
|
||||||
const s = try self.arena.dupe(u8, raw);
|
const s = try self.arena.dupe(u8, raw);
|
||||||
return .{ .string = s };
|
return .{ .string = s };
|
||||||
}
|
}
|
||||||
|
|
||||||
fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const start = cur.i;
|
const start = current.i;
|
||||||
const len = try cur.pkgLen();
|
const len = try current.packageLength();
|
||||||
const end = @min(start + len, cur.b.len);
|
const end = @min(start + len, current.b.len);
|
||||||
const size = try self.evalInt(cur, frame);
|
const size = try self.evaluateInteger(current, frame);
|
||||||
const data = cur.b[@min(cur.i, end)..end];
|
const data = current.b[@min(current.i, end)..end];
|
||||||
const buf = try self.arena.alloc(u8, @intCast(size));
|
const bytes = try self.arena.alloc(u8, @intCast(size));
|
||||||
@memset(buf, 0);
|
@memset(bytes, 0);
|
||||||
@memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]);
|
@memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]);
|
||||||
cur.i = end;
|
current.i = end;
|
||||||
return .{ .buffer = buf };
|
return .{ .buffer = bytes };
|
||||||
}
|
}
|
||||||
|
|
||||||
fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||||
const start = cur.i;
|
const start = current.i;
|
||||||
const len = try cur.pkgLen();
|
const len = try current.packageLength();
|
||||||
const end = @min(start + len, cur.b.len);
|
const end = @min(start + len, current.b.len);
|
||||||
const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte();
|
const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte();
|
||||||
const elems = try self.arena.alloc(Object, count);
|
const elems = try self.arena.alloc(Object, count);
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame);
|
while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame);
|
||||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||||
cur.i = end;
|
current.i = end;
|
||||||
return .{ .package = elems };
|
return .{ .package = elems };
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- operators ----------------------------------------------------------
|
// --- operators ----------------------------------------------------------
|
||||||
|
|
||||||
const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||||
|
|
||||||
fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object {
|
fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object {
|
||||||
const a = try self.evalInt(cur, frame);
|
const a = try self.evaluateInteger(current, frame);
|
||||||
const b = try self.evalInt(cur, frame);
|
const b = try self.evaluateInteger(current, frame);
|
||||||
const r: u64 = switch (kind) {
|
const r: u64 = switch (kind) {
|
||||||
.add => a +% b,
|
.add => a +% b,
|
||||||
.sub => a -% b,
|
.sub => a -% b,
|
||||||
@@ -367,24 +366,24 @@ pub const Interp = struct {
|
|||||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||||
};
|
};
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
return .{ .integer = r };
|
return .{ .integer = r };
|
||||||
}
|
}
|
||||||
|
|
||||||
fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const a = try self.evalInt(cur, frame);
|
const a = try self.evaluateInteger(current, frame);
|
||||||
const b = try self.evalInt(cur, frame);
|
const b = try self.evaluateInteger(current, frame);
|
||||||
if (b == 0) return error.DivByZero;
|
if (b == 0) return error.DivByZero;
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target
|
try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target
|
try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target
|
||||||
return .{ .integer = a / b };
|
return .{ .integer = a / b };
|
||||||
}
|
}
|
||||||
|
|
||||||
const LogicOp = enum { land, lor, eq, gt, lt };
|
const LogicOperation = enum { land, lor, eq, gt, lt };
|
||||||
|
|
||||||
fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object {
|
fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object {
|
||||||
const a = try self.evalInt(cur, frame);
|
const a = try self.evaluateInteger(current, frame);
|
||||||
const b = try self.evalInt(cur, frame);
|
const b = try self.evaluateInteger(current, frame);
|
||||||
const r = switch (kind) {
|
const r = switch (kind) {
|
||||||
.land => a != 0 and b != 0,
|
.land => a != 0 and b != 0,
|
||||||
.lor => a != 0 or b != 0,
|
.lor => a != 0 or b != 0,
|
||||||
@@ -395,48 +394,48 @@ pub const Interp = struct {
|
|||||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||||
}
|
}
|
||||||
|
|
||||||
fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
// 0x92 0x93/94/95 are the compound comparisons.
|
// 0x92 0x93/94/95 are the compound comparisons.
|
||||||
const b = cur.peek() orelse return error.Truncated;
|
const b = current.peek() orelse return error.Truncated;
|
||||||
switch (b) {
|
switch (b) {
|
||||||
op.lnot.not_equal => {
|
opcode.lnot.not_equal => {
|
||||||
cur.i += 1;
|
current.i += 1;
|
||||||
const x = try self.evalInt(cur, frame);
|
const x = try self.evaluateInteger(current, frame);
|
||||||
const y = try self.evalInt(cur, frame);
|
const y = try self.evaluateInteger(current, frame);
|
||||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||||
},
|
},
|
||||||
op.lnot.less_equal => {
|
opcode.lnot.less_equal => {
|
||||||
cur.i += 1;
|
current.i += 1;
|
||||||
const x = try self.evalInt(cur, frame);
|
const x = try self.evaluateInteger(current, frame);
|
||||||
const y = try self.evalInt(cur, frame);
|
const y = try self.evaluateInteger(current, frame);
|
||||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||||
},
|
},
|
||||||
op.lnot.greater_equal => {
|
opcode.lnot.greater_equal => {
|
||||||
cur.i += 1;
|
current.i += 1;
|
||||||
const x = try self.evalInt(cur, frame);
|
const x = try self.evaluateInteger(current, frame);
|
||||||
const y = try self.evalInt(cur, frame);
|
const y = try self.evaluateInteger(current, frame);
|
||||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||||
},
|
},
|
||||||
else => {
|
else => {
|
||||||
const x = try self.evalInt(cur, frame);
|
const x = try self.evaluateInteger(current, frame);
|
||||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||||
// Operand is a SuperName that is both read and written.
|
// Operand is a SuperName that is both read and written.
|
||||||
const save = cur.i;
|
const save = current.i;
|
||||||
const cur_val = try self.term(cur, frame);
|
const current_value = try self.term(current, frame);
|
||||||
const v = try cur_val.asInt();
|
const v = try current_value.asInteger();
|
||||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||||
var tcur = Cursor{ .b = cur.b, .i = save };
|
var tcur = Cursor{ .b = current.b, .i = save };
|
||||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||||
return .{ .integer = r };
|
return .{ .integer = r };
|
||||||
}
|
}
|
||||||
|
|
||||||
fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const o = try self.term(cur, frame);
|
const o = try self.term(current, frame);
|
||||||
return .{ .integer = switch (o) {
|
return .{ .integer = switch (o) {
|
||||||
.buffer => |b| b.len,
|
.buffer => |b| b.len,
|
||||||
.string => |s| s.len,
|
.string => |s| s.len,
|
||||||
@@ -445,29 +444,29 @@ pub const Interp = struct {
|
|||||||
} };
|
} };
|
||||||
}
|
}
|
||||||
|
|
||||||
fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const o = try self.term(cur, frame);
|
const o = try self.term(current, frame);
|
||||||
try self.storeTarget(cur, frame, o);
|
try self.storeTarget(current, frame, o);
|
||||||
return o;
|
return o;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const src = try self.term(cur, frame);
|
const source = try self.term(current, frame);
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
// Optional target (a reference); we don't materialise references, so store
|
// Optional target (a reference); we don't materialise references, so store
|
||||||
// the indexed value if a target is present.
|
// the indexed value if a target is present.
|
||||||
const val: Object = switch (src) {
|
const value: Object = switch (source) {
|
||||||
.buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 },
|
.buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 },
|
||||||
.package => |p| if (idx < p.len) p[idx] else .uninitialized,
|
.package => |p| if (element_index < p.len) p[element_index] else .uninitialized,
|
||||||
.string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 },
|
.string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 },
|
||||||
else => .uninitialized,
|
else => .uninitialized,
|
||||||
};
|
};
|
||||||
try self.storeTarget(cur, frame, val);
|
try self.storeTarget(current, frame, value);
|
||||||
return val;
|
return value;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const o = try self.term(cur, frame);
|
const o = try self.term(current, frame);
|
||||||
return switch (o) {
|
return switch (o) {
|
||||||
.reference => |n| self.invoke(n, &.{}),
|
.reference => |n| self.invoke(n, &.{}),
|
||||||
else => o,
|
else => o,
|
||||||
@@ -476,101 +475,101 @@ pub const Interp = struct {
|
|||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const start = cur.i;
|
const start = current.i;
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
const cond = try self.evalInt(cur, frame);
|
const cond = try self.evaluateInteger(current, frame);
|
||||||
if (cond != 0) {
|
if (cond != 0) {
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = cur.i };
|
var body = Cursor{ .b = current.b[0..end], .i = current.i };
|
||||||
try self.execList(&body, frame);
|
try self.executeList(&body, frame);
|
||||||
cur.i = end;
|
current.i = end;
|
||||||
// Skip a trailing Else.
|
// Skip a trailing Else.
|
||||||
if (cur.peek() == op.else_op) {
|
if (current.peek() == opcode.else_opcode) {
|
||||||
cur.i += 1;
|
current.i += 1;
|
||||||
const es = cur.i;
|
const es = current.i;
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
cur.i = ee;
|
current.i = ee;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
cur.i = end;
|
current.i = end;
|
||||||
if (cur.peek() == op.else_op) {
|
if (current.peek() == opcode.else_opcode) {
|
||||||
cur.i += 1;
|
current.i += 1;
|
||||||
const es = cur.i;
|
const es = current.i;
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
var body = Cursor{ .b = cur.b[0..ee], .i = cur.i };
|
var body = Cursor{ .b = current.b[0..ee], .i = current.i };
|
||||||
try self.execList(&body, frame);
|
try self.executeList(&body, frame);
|
||||||
cur.i = ee;
|
current.i = ee;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return .uninitialized;
|
return .uninitialized;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const start = cur.i;
|
const start = current.i;
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
const pred_at = cur.i;
|
const pred_at = current.i;
|
||||||
var guard: usize = 0;
|
var guard: usize = 0;
|
||||||
while (guard < 100_000) : (guard += 1) {
|
while (guard < 100_000) : (guard += 1) {
|
||||||
var pc = Cursor{ .b = cur.b[0..end], .i = pred_at };
|
var pc = Cursor{ .b = current.b[0..end], .i = pred_at };
|
||||||
const cond = try self.evalInt(&pc, frame);
|
const cond = try self.evaluateInteger(&pc, frame);
|
||||||
if (cond == 0) break;
|
if (cond == 0) break;
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = pc.i };
|
var body = Cursor{ .b = current.b[0..end], .i = pc.i };
|
||||||
try self.execList(&body, frame);
|
try self.executeList(&body, frame);
|
||||||
if (frame.returned) break;
|
if (frame.returned) break;
|
||||||
if (frame.broke) {
|
if (frame.broke) {
|
||||||
frame.broke = false;
|
frame.broke = false;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
cur.i = end;
|
current.i = end;
|
||||||
return .uninitialized;
|
return .uninitialized;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- store --------------------------------------------------------------
|
// --- store --------------------------------------------------------------
|
||||||
|
|
||||||
fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const value = try self.term(cur, frame);
|
const value = try self.term(current, frame);
|
||||||
try self.storeInto(cur, frame, value);
|
try self.storeInto(current, frame, value);
|
||||||
return value;
|
return value;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A Store *target* that may be NullName (no store).
|
/// A Store *target* that may be NullName (no store).
|
||||||
fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
if (cur.peek() == 0x00) {
|
if (current.peek() == 0x00) {
|
||||||
cur.i += 1; // NullName
|
current.i += 1; // NullName
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
try self.storeInto(cur, frame, value);
|
try self.storeInto(current, frame, value);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
if (isNameStart(lead)) {
|
if (isNameStart(lead)) {
|
||||||
const np = try cur.nameString();
|
const name_path = try current.nameString();
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return;
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return;
|
||||||
if (self.fields.get(node)) |bf| {
|
if (self.fields.get(node)) |buffer_field| {
|
||||||
try self.writeBufField(bf, try value.asInt());
|
try self.writeBufferField(buffer_field, try value.asInteger());
|
||||||
} else if (node.kind == .field) {
|
} else if (node.kind == .field) {
|
||||||
try self.writeField(node, try value.asInt());
|
try self.writeField(node, try value.asInteger());
|
||||||
} else {
|
} else {
|
||||||
try self.dyn.put(self.arena, node, value);
|
try self.dynamic_overrides.put(self.arena, node, value);
|
||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
_ = try cur.byte();
|
_ = try current.byte();
|
||||||
switch (lead) {
|
switch (lead) {
|
||||||
0x00 => {}, // NullName
|
0x00 => {}, // NullName
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value,
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value,
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value,
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value,
|
||||||
op.index_op => {
|
opcode.index_opcode => {
|
||||||
const src = try self.term(cur, frame);
|
const source = try self.term(current, frame);
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
switch (src) {
|
switch (source) {
|
||||||
.buffer => |b| if (idx < b.len) {
|
.buffer => |b| if (element_index < b.len) {
|
||||||
b[idx] = @truncate(try value.asInt());
|
b[element_index] = @truncate(try value.asInteger());
|
||||||
},
|
},
|
||||||
.package => |p| if (idx < p.len) {
|
.package => |p| if (element_index < p.len) {
|
||||||
p[idx] = value;
|
p[element_index] = value;
|
||||||
},
|
},
|
||||||
else => {},
|
else => {},
|
||||||
}
|
}
|
||||||
@@ -581,144 +580,144 @@ pub const Interp = struct {
|
|||||||
|
|
||||||
// --- CreateField (buffer patching) --------------------------------------
|
// --- CreateField (buffer patching) --------------------------------------
|
||||||
|
|
||||||
fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||||
const src = try self.term(cur, frame); // source buffer (as a reference or value)
|
const source = try self.term(current, frame); // source buffer (as a reference or value)
|
||||||
const bit_index = try self.evalInt(cur, frame);
|
const bit_index = try self.evaluateInteger(current, frame);
|
||||||
const np = try cur.nameString();
|
const name_path = try current.nameString();
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized;
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized;
|
||||||
|
|
||||||
// Bind the new name to the source buffer's node so stores land in it.
|
// Bind the new name to the source buffer's node so stores land in it.
|
||||||
const buf_node: *Node = switch (src) {
|
const buffer_node: *Node = switch (source) {
|
||||||
.reference => |n| n,
|
.reference => |n| n,
|
||||||
else => return .uninitialized,
|
else => return .uninitialized,
|
||||||
};
|
};
|
||||||
// Materialise the buffer into `dyn` so patches persist and are returned.
|
// Materialise the buffer into `dynamic_overrides` so patches persist and are returned.
|
||||||
if (self.dyn.get(buf_node) == null) {
|
if (self.dynamic_overrides.get(buffer_node) == null) {
|
||||||
const val = try self.invoke(buf_node, &.{});
|
const value = try self.invoke(buffer_node, &.{});
|
||||||
try self.dyn.put(self.arena, buf_node, val);
|
try self.dynamic_overrides.put(self.arena, buffer_node, value);
|
||||||
}
|
}
|
||||||
const byte_off: usize = @intCast(bit_index / 8);
|
const byte_off: usize = @intCast(bit_index / 8);
|
||||||
try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width });
|
try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||||
return .uninitialized;
|
return .uninitialized;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void {
|
fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void {
|
||||||
const obj = self.dyn.get(bf.buf) orelse return;
|
const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return;
|
||||||
const buf = switch (obj) {
|
const bytes = switch (obj) {
|
||||||
.buffer => |b| b,
|
.buffer => |b| b,
|
||||||
else => return,
|
else => return,
|
||||||
};
|
};
|
||||||
const nbytes = (bf.bit_width + 7) / 8;
|
const byte_count = (buffer_field.bit_width + 7) / 8;
|
||||||
var k: usize = 0;
|
var k: usize = 0;
|
||||||
while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) {
|
while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) {
|
||||||
buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- OperationRegion field access ---------------------------------------
|
// --- OperationRegion field access ---------------------------------------
|
||||||
|
|
||||||
fn readField(self: *Interp, field: *Node) Error!u64 {
|
fn readField(self: *Interpreter, field: *Node) Error!u64 {
|
||||||
const region = field.region orelse return error.Unsupported;
|
const region = field.region orelse return error.Unsupported;
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
const base = try self.regionBase(region);
|
const base = try self.regionBase(region);
|
||||||
const start_byte = base + field.bit_offset / 8;
|
const start_byte = base + field.bit_offset / 8;
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
const nbytes = (total + 7) / 8;
|
const byte_count = (total + 7) / 8;
|
||||||
var raw: u128 = 0;
|
var raw: u128 = 0;
|
||||||
var k: usize = 0;
|
var k: usize = 0;
|
||||||
while (k < nbytes) : (k += 1) {
|
while (k < byte_count) : (k += 1) {
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
}
|
}
|
||||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||||
return @truncate(masked);
|
return @truncate(masked);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn writeField(self: *Interp, field: *Node, value: u64) Error!void {
|
fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void {
|
||||||
const region = field.region orelse return error.Unsupported;
|
const region = field.region orelse return error.Unsupported;
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
const base = try self.regionBase(region);
|
const base = try self.regionBase(region);
|
||||||
const start_byte = base + field.bit_offset / 8;
|
const start_byte = base + field.bit_offset / 8;
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
const nbytes = (total + 7) / 8;
|
const byte_count = (total + 7) / 8;
|
||||||
// Read-modify-write byte by byte.
|
// Read-modify-write byte by byte.
|
||||||
var raw: u128 = 0;
|
var raw: u128 = 0;
|
||||||
var k: usize = 0;
|
var k: usize = 0;
|
||||||
while (k < nbytes) : (k += 1) {
|
while (k < byte_count) : (k += 1) {
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
}
|
}
|
||||||
const mask = bitMask(field.bit_width) << shift;
|
const mask = bitMask(field.bit_width) << shift;
|
||||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||||
k = 0;
|
k = 0;
|
||||||
while (k < nbytes) : (k += 1) {
|
while (k < byte_count) : (k += 1) {
|
||||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn regionBase(self: *Interp, region: *Node) Error!u64 {
|
fn regionBase(self: *Interpreter, region: *Node) Error!u64 {
|
||||||
var cur = Cursor{ .b = region.region_offset_aml };
|
var current = Cursor{ .b = region.region_offset_aml };
|
||||||
var frame = Frame{ .scope = region.parent orelse self.ns.root };
|
var frame = Frame{ .scope = region.parent orelse self.namespace.root };
|
||||||
return (try self.term(&cur, &frame)).asInt();
|
return (try self.term(¤t, &frame)).asInteger();
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 {
|
||||||
switch (space) {
|
switch (space) {
|
||||||
0 => { // SystemMemory
|
0 => { // SystemMemory
|
||||||
const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true);
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
const p: *align(1) const volatile u8 = @ptrFromInt(virt + (addr & 0xFFF));
|
const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
return p.*;
|
return p.*;
|
||||||
},
|
},
|
||||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO
|
||||||
else => return error.Unsupported,
|
else => return error.Unsupported,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void {
|
||||||
switch (space) {
|
switch (space) {
|
||||||
0 => {
|
0 => {
|
||||||
const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true);
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
const p: *align(1) volatile u8 = @ptrFromInt(virt + (addr & 0xFFF));
|
const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
p.* = value;
|
p.* = value;
|
||||||
},
|
},
|
||||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value),
|
||||||
else => return error.Unsupported,
|
else => return error.Unsupported,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- extended opcodes ---------------------------------------------------
|
// --- extended opcodes ---------------------------------------------------
|
||||||
|
|
||||||
fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
const e = try cur.byte();
|
const e = try current.byte();
|
||||||
switch (e) {
|
switch (e) {
|
||||||
op.ext.debug => return .uninitialized,
|
opcode.extended.debug => return .uninitialized,
|
||||||
op.ext.revision => return .{ .integer = 2 },
|
opcode.extended.revision => return .{ .integer = 2 },
|
||||||
op.ext.timer => return .{ .integer = 0 },
|
opcode.extended.timer => return .{ .integer = 0 },
|
||||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||||
op.ext.acquire => {
|
opcode.extended.acquire => {
|
||||||
_ = try self.term(cur, frame); // mutex SuperName
|
_ = try self.term(current, frame); // mutex SuperName
|
||||||
_ = try cur.take(2); // timeout
|
_ = try current.take(2); // timeout
|
||||||
return .{ .integer = 0 }; // acquired
|
return .{ .integer = 0 }; // acquired
|
||||||
},
|
},
|
||||||
op.ext.release, op.ext.reset, op.ext.signal => {
|
opcode.extended.release, opcode.extended.reset, opcode.extended.signal => {
|
||||||
_ = try self.term(cur, frame);
|
_ = try self.term(current, frame);
|
||||||
return .uninitialized;
|
return .uninitialized;
|
||||||
},
|
},
|
||||||
op.ext.wait => {
|
opcode.extended.wait => {
|
||||||
_ = try self.term(cur, frame);
|
_ = try self.term(current, frame);
|
||||||
_ = try self.term(cur, frame);
|
_ = try self.term(current, frame);
|
||||||
return .{ .integer = 0 };
|
return .{ .integer = 0 };
|
||||||
},
|
},
|
||||||
op.ext.sleep, op.ext.stall => {
|
opcode.extended.sleep, opcode.extended.stall => {
|
||||||
_ = try self.term(cur, frame);
|
_ = try self.term(current, frame);
|
||||||
return .uninitialized;
|
return .uninitialized;
|
||||||
},
|
},
|
||||||
else => return error.Unsupported,
|
else => return error.Unsupported,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 {
|
fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 {
|
||||||
return (try self.term(cur, frame)).asInt();
|
return (try self.term(current, frame)).asInteger();
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -728,10 +727,10 @@ fn bitMask(width: u32) u128 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
fn isNameStart(b: u8) bool {
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
b == op.name_char_underscore or
|
b == opcode.name_char_underscore or
|
||||||
b == op.root_char or
|
b == opcode.root_char or
|
||||||
b == op.parent_prefix_char or
|
b == opcode.parent_prefix_char or
|
||||||
b == op.dual_name_prefix or
|
b == opcode.dual_name_prefix or
|
||||||
b == op.multi_name_prefix;
|
b == opcode.multi_name_prefix;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -18,7 +18,7 @@ pub const NodeKind = enum {
|
|||||||
mutex,
|
mutex,
|
||||||
event,
|
event,
|
||||||
processor,
|
processor,
|
||||||
power_res,
|
power_resource,
|
||||||
thermal_zone,
|
thermal_zone,
|
||||||
alias,
|
alias,
|
||||||
external,
|
external,
|
||||||
@@ -28,12 +28,12 @@ pub const NodeKind = enum {
|
|||||||
pub const Node = struct {
|
pub const Node = struct {
|
||||||
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
||||||
/// all-zero.
|
/// all-zero.
|
||||||
seg: [4]u8 = .{ 0, 0, 0, 0 },
|
segment: [4]u8 = .{ 0, 0, 0, 0 },
|
||||||
kind: NodeKind = .other,
|
kind: NodeKind = .other,
|
||||||
/// For Method / External: the declared argument count (0..7). Used to resolve
|
/// For Method / External: the declared argument count (0..7). Used to resolve
|
||||||
/// how many TermArgs a method invocation consumes.
|
/// how many TermArgs a method invocation consumes.
|
||||||
arg_count: u8 = 0,
|
arg_count: u8 = 0,
|
||||||
/// For Name: the AML bytes of its DataRefObject (so a value like a sleep
|
/// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep
|
||||||
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
||||||
/// interpreted on demand by the evaluator. Empty otherwise.
|
/// interpreted on demand by the evaluator. Empty otherwise.
|
||||||
value: []const u8 = &.{},
|
value: []const u8 = &.{},
|
||||||
@@ -78,49 +78,49 @@ pub const Namespace = struct {
|
|||||||
return self.root.subtreeCount();
|
return self.root.subtreeCount();
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findChild(parent: *Node, seg: [4]u8) ?*Node {
|
fn findChild(parent: *Node, segment: [4]u8) ?*Node {
|
||||||
var c = parent.first_child;
|
var c = parent.first_child;
|
||||||
while (c) |child| : (c = child.next_sibling) {
|
while (c) |child| : (c = child.next_sibling) {
|
||||||
if (std.mem.eql(u8, &child.seg, &seg)) return child;
|
if (std.mem.eql(u8, &child.segment, &segment)) return child;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The direct child of `node` named `seg`, or null. Unlike `resolve`, this does
|
/// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does
|
||||||
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
||||||
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
||||||
pub fn childOf(node: *Node, seg: [4]u8) ?*Node {
|
pub fn childOf(node: *Node, segment: [4]u8) ?*Node {
|
||||||
return findChild(node, seg);
|
return findChild(node, segment);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn newChild(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
const n = try self.allocator.create(Node);
|
const n = try self.allocator.create(Node);
|
||||||
n.* = .{ .seg = seg, .kind = kind, .parent = parent };
|
n.* = .{ .segment = segment, .kind = kind, .parent = parent };
|
||||||
// Append at the tail so a dump reads in declaration order.
|
// Append at the tail so a dump reads in declaration order.
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = n;
|
parent.first_child = n;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = n;
|
current.next_sibling = n;
|
||||||
}
|
}
|
||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a Field unit node directly under `scope` (field units live in the
|
/// Create a Field unit node directly under `scope` (field units live in the
|
||||||
/// scope of the Field/IndexField/BankField, not under the region).
|
/// scope of the Field/IndexField/BankField, not under the region).
|
||||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, seg: [4]u8) !*Node {
|
pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node {
|
||||||
return self.findOrCreate(scope, seg, .field);
|
return self.findOrCreate(scope, segment, .field);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findOrCreate(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
if (findChild(parent, seg)) |existing| {
|
if (findChild(parent, segment)) |existing| {
|
||||||
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
||||||
// specific kind rather than downgrading to a plain scope.
|
// specific kind rather than downgrading to a plain scope.
|
||||||
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
||||||
return existing;
|
return existing;
|
||||||
}
|
}
|
||||||
return self.newChild(parent, seg, kind);
|
return self.newChild(parent, segment, kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The node a definition's NameString names, creating any intermediate scopes.
|
/// The node a definition's NameString names, creating any intermediate scopes.
|
||||||
@@ -131,16 +131,16 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
kind: NodeKind,
|
kind: NodeKind,
|
||||||
) !*Node {
|
) !*Node {
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
if (segs.len == 0) return base;
|
if (segments.len == 0) return base;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i + 1 < segs.len) : (i += 1) {
|
while (i + 1 < segments.len) : (i += 1) {
|
||||||
base = try self.findOrCreate(base, segs[i], .scope);
|
base = try self.findOrCreate(base, segments[i], .scope);
|
||||||
}
|
}
|
||||||
return self.findOrCreate(base, segs[segs.len - 1], kind);
|
return self.findOrCreate(base, segments[segments.len - 1], kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a NameString *reference* to an existing node, or null. A single
|
/// Resolve a NameString *reference* to an existing node, or null. A single
|
||||||
@@ -151,22 +151,22 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
) ?*Node {
|
) ?*Node {
|
||||||
if (segs.len == 0) return null;
|
if (segments.len == 0) return null;
|
||||||
|
|
||||||
if (!rooted and parents == 0 and segs.len == 1) {
|
if (!rooted and parents == 0 and segments.len == 1) {
|
||||||
// Search rule: this scope, then each ancestor up to the root.
|
// Search rule: this scope, then each ancestor up to the root.
|
||||||
var scope: ?*Node = current;
|
var scope: ?*Node = current;
|
||||||
while (scope) |s| : (scope = s.parent) {
|
while (scope) |s| : (scope = s.parent) {
|
||||||
if (findChild(s, segs[0])) |n| return n;
|
if (findChild(s, segments[0])) |n| return n;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
for (segs) |seg| {
|
for (segments) |segment| {
|
||||||
base = findChild(base, seg) orelse return null;
|
base = findChild(base, segment) orelse return null;
|
||||||
}
|
}
|
||||||
return base;
|
return base;
|
||||||
}
|
}
|
||||||
|
|||||||
+76
-76
@@ -2,28 +2,28 @@
|
|||||||
//!
|
//!
|
||||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||||
//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`).
|
//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`).
|
||||||
|
|
||||||
// --- name / path characters -------------------------------------------------
|
// --- name / path characters -------------------------------------------------
|
||||||
pub const zero_op = 0x00;
|
pub const zero_opcode = 0x00;
|
||||||
pub const one_op = 0x01;
|
pub const one_opcode = 0x01;
|
||||||
pub const alias_op = 0x06;
|
pub const alias_opcode = 0x06;
|
||||||
pub const name_op = 0x08;
|
pub const name_opcode = 0x08;
|
||||||
pub const byte_prefix = 0x0A;
|
pub const byte_prefix = 0x0A;
|
||||||
pub const word_prefix = 0x0B;
|
pub const word_prefix = 0x0B;
|
||||||
pub const dword_prefix = 0x0C;
|
pub const dword_prefix = 0x0C;
|
||||||
pub const string_prefix = 0x0D;
|
pub const string_prefix = 0x0D;
|
||||||
pub const qword_prefix = 0x0E;
|
pub const qword_prefix = 0x0E;
|
||||||
pub const scope_op = 0x10;
|
pub const scope_opcode = 0x10;
|
||||||
pub const buffer_op = 0x11;
|
pub const buffer_opcode = 0x11;
|
||||||
pub const package_op = 0x12;
|
pub const package_opcode = 0x12;
|
||||||
pub const var_package_op = 0x13;
|
pub const var_package_opcode = 0x13;
|
||||||
pub const method_op = 0x14;
|
pub const method_opcode = 0x14;
|
||||||
pub const external_op = 0x15;
|
pub const external_opcode = 0x15;
|
||||||
|
|
||||||
pub const dual_name_prefix = 0x2E;
|
pub const dual_name_prefix = 0x2E;
|
||||||
pub const multi_name_prefix = 0x2F;
|
pub const multi_name_prefix = 0x2F;
|
||||||
pub const ext_op_prefix = 0x5B;
|
pub const extended_opcode_prefix = 0x5B;
|
||||||
pub const root_char = 0x5C;
|
pub const root_char = 0x5C;
|
||||||
pub const parent_prefix_char = 0x5E;
|
pub const parent_prefix_char = 0x5E;
|
||||||
pub const name_char_underscore = 0x5F;
|
pub const name_char_underscore = 0x5F;
|
||||||
@@ -34,80 +34,80 @@ pub const name_char_start = 0x41; // 'A'
|
|||||||
pub const name_char_end = 0x5A; // 'Z'
|
pub const name_char_end = 0x5A; // 'Z'
|
||||||
|
|
||||||
// --- locals / args ----------------------------------------------------------
|
// --- locals / args ----------------------------------------------------------
|
||||||
pub const local0_op = 0x60;
|
pub const local0_opcode = 0x60;
|
||||||
pub const local7_op = 0x67;
|
pub const local7_opcode = 0x67;
|
||||||
pub const arg0_op = 0x68;
|
pub const arg0_opcode = 0x68;
|
||||||
pub const arg6_op = 0x6E;
|
pub const arg6_opcode = 0x6E;
|
||||||
|
|
||||||
// --- store / references / arithmetic ---------------------------------------
|
// --- store / references / arithmetic ---------------------------------------
|
||||||
pub const store_op = 0x70;
|
pub const store_opcode = 0x70;
|
||||||
pub const ref_of_op = 0x71;
|
pub const ref_of_opcode = 0x71;
|
||||||
pub const add_op = 0x72;
|
pub const add_opcode = 0x72;
|
||||||
pub const concat_op = 0x73;
|
pub const concat_opcode = 0x73;
|
||||||
pub const subtract_op = 0x74;
|
pub const subtract_opcode = 0x74;
|
||||||
pub const increment_op = 0x75;
|
pub const increment_opcode = 0x75;
|
||||||
pub const decrement_op = 0x76;
|
pub const decrement_opcode = 0x76;
|
||||||
pub const multiply_op = 0x77;
|
pub const multiply_opcode = 0x77;
|
||||||
pub const divide_op = 0x78;
|
pub const divide_opcode = 0x78;
|
||||||
pub const shift_left_op = 0x79;
|
pub const shift_left_opcode = 0x79;
|
||||||
pub const shift_right_op = 0x7A;
|
pub const shift_right_opcode = 0x7A;
|
||||||
pub const and_op = 0x7B;
|
pub const and_opcode = 0x7B;
|
||||||
pub const nand_op = 0x7C;
|
pub const nand_opcode = 0x7C;
|
||||||
pub const or_op = 0x7D;
|
pub const or_opcode = 0x7D;
|
||||||
pub const nor_op = 0x7E;
|
pub const nor_opcode = 0x7E;
|
||||||
pub const xor_op = 0x7F;
|
pub const xor_opcode = 0x7F;
|
||||||
pub const not_op = 0x80;
|
pub const not_opcode = 0x80;
|
||||||
pub const find_set_left_bit_op = 0x81;
|
pub const find_set_left_bit_opcode = 0x81;
|
||||||
pub const find_set_right_bit_op = 0x82;
|
pub const find_set_right_bit_opcode = 0x82;
|
||||||
pub const deref_of_op = 0x83;
|
pub const dereference_of_opcode = 0x83;
|
||||||
pub const concat_res_op = 0x84;
|
pub const concat_resource_opcode = 0x84;
|
||||||
pub const mod_op = 0x85;
|
pub const mod_opcode = 0x85;
|
||||||
pub const notify_op = 0x86;
|
pub const notify_opcode = 0x86;
|
||||||
pub const size_of_op = 0x87;
|
pub const size_of_opcode = 0x87;
|
||||||
pub const index_op = 0x88;
|
pub const index_opcode = 0x88;
|
||||||
pub const match_op = 0x89;
|
pub const match_opcode = 0x89;
|
||||||
pub const create_dword_field_op = 0x8A;
|
pub const create_dword_field_opcode = 0x8A;
|
||||||
pub const create_word_field_op = 0x8B;
|
pub const create_word_field_opcode = 0x8B;
|
||||||
pub const create_byte_field_op = 0x8C;
|
pub const create_byte_field_opcode = 0x8C;
|
||||||
pub const create_bit_field_op = 0x8D;
|
pub const create_bit_field_opcode = 0x8D;
|
||||||
pub const object_type_op = 0x8E;
|
pub const object_type_opcode = 0x8E;
|
||||||
pub const create_qword_field_op = 0x8F;
|
pub const create_qword_field_opcode = 0x8F;
|
||||||
|
|
||||||
pub const land_op = 0x90;
|
pub const land_opcode = 0x90;
|
||||||
pub const lor_op = 0x91;
|
pub const lor_opcode = 0x91;
|
||||||
pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`)
|
pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`)
|
||||||
pub const lequal_op = 0x93;
|
pub const lequal_opcode = 0x93;
|
||||||
pub const lgreater_op = 0x94;
|
pub const lgreater_opcode = 0x94;
|
||||||
pub const lless_op = 0x95;
|
pub const lless_opcode = 0x95;
|
||||||
pub const to_buffer_op = 0x96;
|
pub const to_buffer_opcode = 0x96;
|
||||||
pub const to_decimal_string_op = 0x97;
|
pub const to_decimal_string_opcode = 0x97;
|
||||||
pub const to_hex_string_op = 0x98;
|
pub const to_hex_string_opcode = 0x98;
|
||||||
pub const to_integer_op = 0x99;
|
pub const to_integer_opcode = 0x99;
|
||||||
pub const to_string_op = 0x9C;
|
pub const to_string_opcode = 0x9C;
|
||||||
pub const copy_object_op = 0x9D;
|
pub const copy_object_opcode = 0x9D;
|
||||||
pub const mid_op = 0x9E;
|
pub const mid_opcode = 0x9E;
|
||||||
pub const continue_op = 0x9F;
|
pub const continue_opcode = 0x9F;
|
||||||
pub const if_op = 0xA0;
|
pub const if_opcode = 0xA0;
|
||||||
pub const else_op = 0xA1;
|
pub const else_opcode = 0xA1;
|
||||||
pub const while_op = 0xA2;
|
pub const while_opcode = 0xA2;
|
||||||
pub const noop_op = 0xA3;
|
pub const noop_opcode = 0xA3;
|
||||||
pub const return_op = 0xA4;
|
pub const return_opcode = 0xA4;
|
||||||
pub const break_op = 0xA5;
|
pub const break_opcode = 0xA5;
|
||||||
pub const break_point_op = 0xCC;
|
pub const break_point_opcode = 0xCC;
|
||||||
pub const ones_op = 0xFF;
|
pub const ones_opcode = 0xFF;
|
||||||
|
|
||||||
/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes.
|
/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes.
|
||||||
pub const lnot = struct {
|
pub const lnot = struct {
|
||||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B).
|
/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B).
|
||||||
pub const ext = struct {
|
pub const extended = struct {
|
||||||
pub const mutex = 0x01;
|
pub const mutex = 0x01;
|
||||||
pub const event = 0x02;
|
pub const event = 0x02;
|
||||||
pub const cond_ref_of = 0x12;
|
pub const conditional_reference_of = 0x12;
|
||||||
pub const create_field = 0x13;
|
pub const create_field = 0x13;
|
||||||
pub const load_table = 0x1F;
|
pub const load_table = 0x1F;
|
||||||
pub const load = 0x20;
|
pub const load = 0x20;
|
||||||
@@ -125,11 +125,11 @@ pub const ext = struct {
|
|||||||
pub const debug = 0x31;
|
pub const debug = 0x31;
|
||||||
pub const fatal = 0x32;
|
pub const fatal = 0x32;
|
||||||
pub const timer = 0x33;
|
pub const timer = 0x33;
|
||||||
pub const op_region = 0x80;
|
pub const operation_region = 0x80;
|
||||||
pub const field = 0x81;
|
pub const field = 0x81;
|
||||||
pub const device = 0x82;
|
pub const device = 0x82;
|
||||||
pub const processor = 0x83;
|
pub const processor = 0x83;
|
||||||
pub const power_res = 0x84;
|
pub const power_resource = 0x84;
|
||||||
pub const thermal_zone = 0x85;
|
pub const thermal_zone = 0x85;
|
||||||
pub const index_field = 0x86;
|
pub const index_field = 0x86;
|
||||||
pub const bank_field = 0x87;
|
pub const bank_field = 0x87;
|
||||||
|
|||||||
+208
-208
@@ -14,31 +14,31 @@
|
|||||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const op = @import("opcodes.zig");
|
const opcode = @import("opcodes.zig");
|
||||||
const ns = @import("namespace.zig");
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
const Namespace = ns.Namespace;
|
const Node = @import("namespace.zig").Node;
|
||||||
const Node = ns.Node;
|
const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
const max_segs = 64;
|
const maximum_segments = 64;
|
||||||
|
|
||||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||||
/// of 4-byte segments.
|
/// of 4-byte segments.
|
||||||
const NamePath = struct {
|
const NamePath = struct {
|
||||||
rooted: bool = false,
|
rooted: bool = false,
|
||||||
parents: u8 = 0,
|
parents: u8 = 0,
|
||||||
segs: [max_segs][4]u8 = undefined,
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
count: usize = 0,
|
count: usize = 0,
|
||||||
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
return self.segs[0..self.count];
|
return self.segments[0..self.count];
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
pub const Parser = struct {
|
pub const Parser = struct {
|
||||||
aml: []const u8,
|
aml: []const u8,
|
||||||
pos: usize = 0,
|
position: usize = 0,
|
||||||
namespace: *Namespace,
|
namespace: *Namespace,
|
||||||
|
|
||||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||||
@@ -49,29 +49,29 @@ pub const Parser = struct {
|
|||||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||||
pub fn parseAll(self: *Parser) usize {
|
pub fn parseAll(self: *Parser) usize {
|
||||||
self.termList(self.aml.len, self.namespace.root);
|
self.termList(self.aml.len, self.namespace.root);
|
||||||
return self.pos;
|
return self.position;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- cursor primitives --------------------------------------------------
|
// --- cursor primitives --------------------------------------------------
|
||||||
|
|
||||||
fn eof(self: *Parser) bool {
|
fn eof(self: *Parser) bool {
|
||||||
return self.pos >= self.aml.len;
|
return self.position >= self.aml.len;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn peek(self: *Parser) ?u8 {
|
fn peek(self: *Parser) ?u8 {
|
||||||
return if (self.eof()) null else self.aml[self.pos];
|
return if (self.eof()) null else self.aml[self.position];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readByte(self: *Parser) Error!u8 {
|
fn readByte(self: *Parser) Error!u8 {
|
||||||
if (self.eof()) return error.Truncated;
|
if (self.eof()) return error.Truncated;
|
||||||
const b = self.aml[self.pos];
|
const b = self.aml[self.position];
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
return b;
|
return b;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn skip(self: *Parser, n: usize) Error!void {
|
fn skip(self: *Parser, n: usize) Error!void {
|
||||||
if (self.pos + n > self.aml.len) return error.Truncated;
|
if (self.position + n > self.aml.len) return error.Truncated;
|
||||||
self.pos += n;
|
self.position += n;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn skipCString(self: *Parser) Error!void {
|
fn skipCString(self: *Parser) Error!void {
|
||||||
@@ -83,7 +83,7 @@ pub const Parser = struct {
|
|||||||
|
|
||||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||||
/// follow; the value counts from the start of the PkgLength field.
|
/// follow; the value counts from the start of the PkgLength field.
|
||||||
fn readPkgLength(self: *Parser) Error!usize {
|
fn readPackageLength(self: *Parser) Error!usize {
|
||||||
const lead = try self.readByte();
|
const lead = try self.readByte();
|
||||||
const follow: usize = lead >> 6;
|
const follow: usize = lead >> 6;
|
||||||
if (follow == 0) return lead & 0x3F;
|
if (follow == 0) return lead & 0x3F;
|
||||||
@@ -96,49 +96,49 @@ pub const Parser = struct {
|
|||||||
return value;
|
return value;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readNameSeg(self: *Parser) Error![4]u8 {
|
fn readNameSegment(self: *Parser) Error![4]u8 {
|
||||||
if (self.pos + 4 > self.aml.len) return error.Truncated;
|
if (self.position + 4 > self.aml.len) return error.Truncated;
|
||||||
const seg = self.aml[self.pos..][0..4].*;
|
const segment = self.aml[self.position..][0..4].*;
|
||||||
self.pos += 4;
|
self.position += 4;
|
||||||
return seg;
|
return segment;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readNameString(self: *Parser) Error!NamePath {
|
fn readNameString(self: *Parser) Error!NamePath {
|
||||||
var np = NamePath{};
|
var name_path = NamePath{};
|
||||||
// A NameString is either root-anchored or parent-relative, not both.
|
// A NameString is either root-anchored or parent-relative, not both.
|
||||||
if (self.peek() == op.root_char) {
|
if (self.peek() == opcode.root_char) {
|
||||||
np.rooted = true;
|
name_path.rooted = true;
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
} else {
|
} else {
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1;
|
while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
const lead = self.peek() orelse return np;
|
const lead = self.peek() orelse return name_path;
|
||||||
switch (lead) {
|
switch (lead) {
|
||||||
0x00 => self.pos += 1, // NullName
|
0x00 => self.position += 1, // NullName
|
||||||
op.dual_name_prefix => {
|
opcode.dual_name_prefix => {
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
try self.appendSeg(&np);
|
try self.appendSegment(&name_path);
|
||||||
try self.appendSeg(&np);
|
try self.appendSegment(&name_path);
|
||||||
},
|
},
|
||||||
op.multi_name_prefix => {
|
opcode.multi_name_prefix => {
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
const cnt = try self.readByte();
|
const count = try self.readByte();
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < cnt) : (i += 1) try self.appendSeg(&np);
|
while (i < count) : (i += 1) try self.appendSegment(&name_path);
|
||||||
},
|
},
|
||||||
else => {
|
else => {
|
||||||
if (isNameStart(lead)) try self.appendSeg(&np);
|
if (isNameStart(lead)) try self.appendSegment(&name_path);
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
return np;
|
return name_path;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn appendSeg(self: *Parser, np: *NamePath) Error!void {
|
fn appendSegment(self: *Parser, name_path: *NamePath) Error!void {
|
||||||
const seg = try self.readNameSeg();
|
const segment = try self.readNameSegment();
|
||||||
if (np.count < max_segs) {
|
if (name_path.count < maximum_segments) {
|
||||||
np.segs[np.count] = seg;
|
name_path.segments[name_path.count] = segment;
|
||||||
np.count += 1;
|
name_path.count += 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -147,10 +147,10 @@ pub const Parser = struct {
|
|||||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||||
/// the boundary rather than propagating — containment for the rare desync.
|
/// the boundary rather than propagating — containment for the rare desync.
|
||||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||||
while (self.pos < end) {
|
while (self.position < end) {
|
||||||
self.object(scope) catch break;
|
self.object(scope) catch break;
|
||||||
}
|
}
|
||||||
self.pos = end;
|
self.position = end;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||||
@@ -163,66 +163,66 @@ pub const Parser = struct {
|
|||||||
_ = try self.readByte();
|
_ = try self.readByte();
|
||||||
switch (lead) {
|
switch (lead) {
|
||||||
// constants and no-operand statements
|
// constants and no-operand statements
|
||||||
op.zero_op, op.one_op, op.ones_op => {},
|
opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {},
|
||||||
op.noop_op, op.continue_op, op.break_op, op.break_point_op => {},
|
opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {},
|
||||||
op.local0_op...op.local7_op => {},
|
opcode.local0_opcode...opcode.local7_opcode => {},
|
||||||
op.arg0_op...op.arg6_op => {},
|
opcode.arg0_opcode...opcode.arg6_opcode => {},
|
||||||
|
|
||||||
// literal data
|
// literal data
|
||||||
op.byte_prefix => try self.skip(1),
|
opcode.byte_prefix => try self.skip(1),
|
||||||
op.word_prefix => try self.skip(2),
|
opcode.word_prefix => try self.skip(2),
|
||||||
op.dword_prefix => try self.skip(4),
|
opcode.dword_prefix => try self.skip(4),
|
||||||
op.qword_prefix => try self.skip(8),
|
opcode.qword_prefix => try self.skip(8),
|
||||||
op.string_prefix => try self.skipCString(),
|
opcode.string_prefix => try self.skipCString(),
|
||||||
|
|
||||||
// data containers (contents skipped via their PkgLength)
|
// data containers (contents skipped via their PkgLength)
|
||||||
op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(),
|
opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(),
|
||||||
|
|
||||||
// namespace modifiers / named objects
|
// namespace modifiers / named objects
|
||||||
op.name_op => try self.opName(scope),
|
opcode.name_opcode => try self.parseName(scope),
|
||||||
op.alias_op => try self.opAlias(scope),
|
opcode.alias_opcode => try self.parseAlias(scope),
|
||||||
op.scope_op => try self.opScopeLike(scope, .scope),
|
opcode.scope_opcode => try self.parseScopeLike(scope, .scope),
|
||||||
op.method_op => try self.opMethod(scope),
|
opcode.method_opcode => try self.parseMethod(scope),
|
||||||
op.external_op => try self.opExternal(scope),
|
opcode.external_opcode => try self.parseExternal(scope),
|
||||||
op.ext_op_prefix => try self.opExt(scope),
|
opcode.extended_opcode_prefix => try self.parseExtended(scope),
|
||||||
|
|
||||||
// control flow
|
// control flow
|
||||||
op.if_op => try self.opIf(scope),
|
opcode.if_opcode => try self.parseIf(scope),
|
||||||
op.else_op => try self.opElse(scope),
|
opcode.else_opcode => try self.parseElse(scope),
|
||||||
op.while_op => try self.opWhile(scope),
|
opcode.while_opcode => try self.parseWhile(scope),
|
||||||
op.return_op => try self.object(scope),
|
opcode.return_opcode => try self.object(scope),
|
||||||
op.notify_op => try self.args(scope, 2),
|
opcode.notify_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
// stores / references / unary+target
|
// stores / references / unary+target
|
||||||
op.store_op => try self.args(scope, 2),
|
opcode.store_opcode => try self.args(scope, 2),
|
||||||
op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1),
|
opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1),
|
||||||
op.increment_op, op.decrement_op => try self.args(scope, 1),
|
opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1),
|
||||||
op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2),
|
opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2),
|
||||||
op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2),
|
opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2),
|
||||||
op.copy_object_op => try self.args(scope, 2),
|
opcode.copy_object_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
// binary + target
|
// binary + target
|
||||||
op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3),
|
opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3),
|
||||||
op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3),
|
opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3),
|
||||||
op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3),
|
opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3),
|
||||||
op.divide_op => try self.args(scope, 4),
|
opcode.divide_opcode => try self.args(scope, 4),
|
||||||
op.to_string_op => try self.args(scope, 3),
|
opcode.to_string_opcode => try self.args(scope, 3),
|
||||||
op.mid_op => try self.args(scope, 4),
|
opcode.mid_opcode => try self.args(scope, 4),
|
||||||
|
|
||||||
// logical
|
// logical
|
||||||
op.land_op, op.lor_op => try self.args(scope, 2),
|
opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2),
|
||||||
op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2),
|
opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2),
|
||||||
op.lnot_op => try self.opLnot(scope),
|
opcode.lnot_opcode => try self.parseLnot(scope),
|
||||||
|
|
||||||
op.match_op => try self.opMatch(scope),
|
opcode.match_opcode => try self.parseMatch(scope),
|
||||||
|
|
||||||
// CreateXField: <source> <index> NameString
|
// CreateXField: <source> <index> NameString
|
||||||
op.create_dword_field_op,
|
opcode.create_dword_field_opcode,
|
||||||
op.create_word_field_op,
|
opcode.create_word_field_opcode,
|
||||||
op.create_byte_field_op,
|
opcode.create_byte_field_opcode,
|
||||||
op.create_bit_field_op,
|
opcode.create_bit_field_opcode,
|
||||||
op.create_qword_field_op,
|
opcode.create_qword_field_opcode,
|
||||||
=> try self.opCreateField(scope, 2),
|
=> try self.parseCreateField(scope, 2),
|
||||||
|
|
||||||
else => return error.Malformed,
|
else => return error.Malformed,
|
||||||
}
|
}
|
||||||
@@ -237,8 +237,8 @@ pub const Parser = struct {
|
|||||||
/// A NameString in operand/statement position: a method invocation (consuming
|
/// A NameString in operand/statement position: a method invocation (consuming
|
||||||
/// the callee's declared argument count) or a plain name reference.
|
/// the callee's declared argument count) or a plain name reference.
|
||||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| {
|
if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| {
|
||||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||||
try self.args(scope, node.arg_count);
|
try self.args(scope, node.arg_count);
|
||||||
}
|
}
|
||||||
@@ -247,132 +247,132 @@ pub const Parser = struct {
|
|||||||
|
|
||||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||||
/// the contents are pure data, never namespace declarations.
|
/// the contents are pure data, never namespace declarations.
|
||||||
fn skipPkg(self: *Parser) Error!void {
|
fn skipPackage(self: *Parser) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const len = try self.readPkgLength();
|
const len = try self.readPackageLength();
|
||||||
const end = start + len;
|
const end = start + len;
|
||||||
if (end > self.aml.len) return error.Truncated;
|
if (end > self.aml.len) return error.Truncated;
|
||||||
self.pos = end;
|
self.position = end;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- namespace objects --------------------------------------------------
|
// --- namespace objects --------------------------------------------------
|
||||||
|
|
||||||
fn opName(self: *Parser, scope: *Node) Error!void {
|
fn parseName(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
const val_start = self.pos;
|
const value_start = self.position;
|
||||||
try self.object(scope); // the DataRefObject value
|
try self.object(scope); // the DataReferenceObject value
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
node.value = self.aml[val_start..self.pos];
|
node.value = self.aml[value_start..self.position];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opAlias(self: *Parser, scope: *Node) Error!void {
|
fn parseAlias(self: *Parser, scope: *Node) Error!void {
|
||||||
_ = try self.readNameString(); // source
|
_ = try self.readNameString(); // source
|
||||||
const np = try self.readNameString(); // the alias name
|
const name_path = try self.readNameString(); // the alias name
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias);
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opMethod(self: *Parser, scope: *Node) Error!void {
|
fn parseMethod(self: *Parser, scope: *Node) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
const flags = try self.readByte();
|
const flags = try self.readByte();
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method);
|
||||||
node.arg_count = flags & 0x7;
|
node.arg_count = flags & 0x7;
|
||||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||||
// inside a method are created at *runtime*, not at load, so they must not
|
// inside a method are created at *runtime*, not at load, so they must not
|
||||||
// become permanent namespace nodes.
|
// become permanent namespace nodes.
|
||||||
node.value = self.aml[self.pos..@min(end, self.aml.len)];
|
node.value = self.aml[self.position..@min(end, self.aml.len)];
|
||||||
self.pos = end;
|
self.position = end;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opExternal(self: *Parser, scope: *Node) Error!void {
|
fn parseExternal(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
_ = try self.readByte(); // object type
|
_ = try self.readByte(); // object type
|
||||||
const arg_count = try self.readByte();
|
const arg_count = try self.readByte();
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external);
|
||||||
node.arg_count = arg_count;
|
node.arg_count = arg_count;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||||
fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void {
|
fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind);
|
||||||
self.termList(end, node);
|
self.termList(end, node);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opProcessor(self: *Parser, scope: *Node) Error!void {
|
fn parseProcessor(self: *Parser, scope: *Node) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte)
|
try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte)
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor);
|
||||||
self.termList(end, node);
|
self.termList(end, node);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opPowerRes(self: *Parser, scope: *Node) Error!void {
|
fn parsePowerResource(self: *Parser, scope: *Node) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource);
|
||||||
self.termList(end, node);
|
self.termList(end, node);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||||
fn opRegion(self: *Parser, scope: *Node) Error!void {
|
fn parseRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
const space = try self.readByte();
|
const space = try self.readByte();
|
||||||
const off_start = self.pos;
|
const off_start = self.position;
|
||||||
try self.object(scope);
|
try self.object(scope);
|
||||||
const off_end = self.pos;
|
const off_end = self.position;
|
||||||
try self.object(scope);
|
try self.object(scope);
|
||||||
const len_end = self.pos;
|
const len_end = self.position;
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
node.region_space = space;
|
node.region_space = space;
|
||||||
node.region_offset_aml = self.aml[off_start..off_end];
|
node.region_offset_aml = self.aml[off_start..off_end];
|
||||||
node.region_len_aml = self.aml[off_end..len_end];
|
node.region_len_aml = self.aml[off_end..len_end];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opDataRegion(self: *Parser, scope: *Node) Error!void {
|
fn parseDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opMutex(self: *Parser, scope: *Node) Error!void {
|
fn parseMutex(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
try self.skip(1); // sync flags
|
try self.skip(1); // sync flags
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex);
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opEvent(self: *Parser, scope: *Node) Error!void {
|
fn parseEvent(self: *Parser, scope: *Node) Error!void {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event);
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||||
fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||||
try self.args(scope, count);
|
try self.args(scope, count);
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||||
/// Field, the first NameString is the backing region — captured so field units
|
/// Field, the first NameString is the backing region — captured so field units
|
||||||
/// carry a region + bit position the evaluator can read/write.
|
/// carry a region + bit position the evaluator can read/write.
|
||||||
fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
var region: ?*Node = null;
|
var region: ?*Node = null;
|
||||||
var i: u8 = 0;
|
var i: u8 = 0;
|
||||||
while (i < name_strings) : (i += 1) {
|
while (i < name_strings) : (i += 1) {
|
||||||
const np = try self.readNameString();
|
const name_path = try self.readNameString();
|
||||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||||
if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice());
|
if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||||
}
|
}
|
||||||
if (bank) try self.object(scope); // bank value TermArg
|
if (bank) try self.object(scope); // bank value TermArg
|
||||||
const flags = try self.readByte();
|
const flags = try self.readByte();
|
||||||
@@ -382,32 +382,32 @@ pub const Parser = struct {
|
|||||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||||
var bit_offset: u32 = 0;
|
var bit_offset: u32 = 0;
|
||||||
var access = initial_access;
|
var access = initial_access;
|
||||||
while (self.pos < end) {
|
while (self.position < end) {
|
||||||
const lead = self.peek() orelse break;
|
const lead = self.peek() orelse break;
|
||||||
switch (lead) {
|
switch (lead) {
|
||||||
0x00 => { // ReservedField: advances the bit position
|
0x00 => { // ReservedField: advances the bit position
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
const width = self.readPkgLength() catch break;
|
const width = self.readPackageLength() catch break;
|
||||||
bit_offset += @intCast(width);
|
bit_offset += @intCast(width);
|
||||||
},
|
},
|
||||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
const at = self.readByte() catch break;
|
const at = self.readByte() catch break;
|
||||||
self.skip(1) catch break;
|
self.skip(1) catch break;
|
||||||
access = at & 0x0F;
|
access = at & 0x0F;
|
||||||
},
|
},
|
||||||
0x02 => { // ConnectField: NameString | BufferData
|
0x02 => { // ConnectField: NameString | BufferData
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
self.object(scope) catch break;
|
self.object(scope) catch break;
|
||||||
},
|
},
|
||||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
self.skip(3) catch break;
|
self.skip(3) catch break;
|
||||||
},
|
},
|
||||||
else => { // NamedField: NameSeg + PkgLength (bit width)
|
else => { // NamedField: NameSegment + PkgLength (bit width)
|
||||||
const seg = self.readNameSeg() catch break;
|
const segment = self.readNameSegment() catch break;
|
||||||
const width = self.readPkgLength() catch break;
|
const width = self.readPackageLength() catch break;
|
||||||
const unit = self.namespace.newFieldUnit(scope, seg) catch break;
|
const unit = self.namespace.newFieldUnit(scope, segment) catch break;
|
||||||
unit.region = region;
|
unit.region = region;
|
||||||
unit.bit_offset = bit_offset;
|
unit.bit_offset = bit_offset;
|
||||||
unit.bit_width = @intCast(width);
|
unit.bit_width = @intCast(width);
|
||||||
@@ -416,49 +416,49 @@ pub const Parser = struct {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
self.pos = end;
|
self.position = end;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
fn opIf(self: *Parser, scope: *Node) Error!void {
|
fn parseIf(self: *Parser, scope: *Node) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
try self.object(scope); // predicate
|
try self.object(scope); // predicate
|
||||||
self.termList(end, scope);
|
self.termList(end, scope);
|
||||||
if (self.peek() == op.else_op) {
|
if (self.peek() == opcode.else_opcode) {
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
try self.opElse(scope);
|
try self.parseElse(scope);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opElse(self: *Parser, scope: *Node) Error!void {
|
fn parseElse(self: *Parser, scope: *Node) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
self.termList(end, scope);
|
self.termList(end, scope);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opWhile(self: *Parser, scope: *Node) Error!void {
|
fn parseWhile(self: *Parser, scope: *Node) Error!void {
|
||||||
const start = self.pos;
|
const start = self.position;
|
||||||
const end = start + try self.readPkgLength();
|
const end = start + try self.readPackageLength();
|
||||||
try self.object(scope); // predicate
|
try self.object(scope); // predicate
|
||||||
self.termList(end, scope);
|
self.termList(end, scope);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opLnot(self: *Parser, scope: *Node) Error!void {
|
fn parseLnot(self: *Parser, scope: *Node) Error!void {
|
||||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||||
// otherwise it is a plain LNot of one operand.
|
// otherwise it is a plain LNot of one operand.
|
||||||
const b = self.peek() orelse return error.Truncated;
|
const b = self.peek() orelse return error.Truncated;
|
||||||
switch (b) {
|
switch (b) {
|
||||||
op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => {
|
opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => {
|
||||||
self.pos += 1;
|
self.position += 1;
|
||||||
try self.args(scope, 2);
|
try self.args(scope, 2);
|
||||||
},
|
},
|
||||||
else => try self.object(scope),
|
else => try self.object(scope),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn opMatch(self: *Parser, scope: *Node) Error!void {
|
fn parseMatch(self: *Parser, scope: *Node) Error!void {
|
||||||
try self.object(scope); // search package
|
try self.object(scope); // search package
|
||||||
try self.skip(1); // match opcode 1
|
try self.skip(1); // match opcode 1
|
||||||
try self.object(scope); // operand 1
|
try self.object(scope); // operand 1
|
||||||
@@ -469,38 +469,38 @@ pub const Parser = struct {
|
|||||||
|
|
||||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||||
|
|
||||||
fn opExt(self: *Parser, scope: *Node) Error!void {
|
fn parseExtended(self: *Parser, scope: *Node) Error!void {
|
||||||
const e = try self.readByte();
|
const e = try self.readByte();
|
||||||
switch (e) {
|
switch (e) {
|
||||||
op.ext.mutex => try self.opMutex(scope),
|
opcode.extended.mutex => try self.parseMutex(scope),
|
||||||
op.ext.event => try self.opEvent(scope),
|
opcode.extended.event => try self.parseEvent(scope),
|
||||||
op.ext.op_region => try self.opRegion(scope),
|
opcode.extended.operation_region => try self.parseRegion(scope),
|
||||||
op.ext.data_region => try self.opDataRegion(scope),
|
opcode.extended.data_region => try self.parseDataRegion(scope),
|
||||||
op.ext.field => try self.opField(scope, 1, false),
|
opcode.extended.field => try self.parseField(scope, 1, false),
|
||||||
op.ext.index_field => try self.opField(scope, 2, false),
|
opcode.extended.index_field => try self.parseField(scope, 2, false),
|
||||||
op.ext.bank_field => try self.opField(scope, 2, true),
|
opcode.extended.bank_field => try self.parseField(scope, 2, true),
|
||||||
op.ext.device => try self.opScopeLike(scope, .device),
|
opcode.extended.device => try self.parseScopeLike(scope, .device),
|
||||||
op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone),
|
opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone),
|
||||||
op.ext.processor => try self.opProcessor(scope),
|
opcode.extended.processor => try self.parseProcessor(scope),
|
||||||
op.ext.power_res => try self.opPowerRes(scope),
|
opcode.extended.power_resource => try self.parsePowerResource(scope),
|
||||||
|
|
||||||
op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target
|
opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target
|
||||||
op.ext.create_field => try self.opCreateField(scope, 3),
|
opcode.extended.create_field => try self.parseCreateField(scope, 3),
|
||||||
op.ext.load_table => try self.args(scope, 6),
|
opcode.extended.load_table => try self.args(scope, 6),
|
||||||
op.ext.load => try self.args(scope, 2), // NameString, Target
|
opcode.extended.load => try self.args(scope, 2), // NameString, Target
|
||||||
op.ext.stall, op.ext.sleep => try self.args(scope, 1),
|
opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1),
|
||||||
op.ext.acquire => {
|
opcode.extended.acquire => {
|
||||||
try self.object(scope); // mutex SuperName
|
try self.object(scope); // mutex SuperName
|
||||||
try self.skip(2); // timeout WordData
|
try self.skip(2); // timeout WordData
|
||||||
},
|
},
|
||||||
op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1),
|
opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1),
|
||||||
op.ext.wait => try self.args(scope, 2),
|
opcode.extended.wait => try self.args(scope, 2),
|
||||||
op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2),
|
opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2),
|
||||||
op.ext.fatal => {
|
opcode.extended.fatal => {
|
||||||
try self.skip(5); // Type(byte) + Code(dword)
|
try self.skip(5); // Type(byte) + Code(dword)
|
||||||
try self.object(scope); // Arg TermArg
|
try self.object(scope); // Arg TermArg
|
||||||
},
|
},
|
||||||
op.ext.revision, op.ext.debug, op.ext.timer => {},
|
opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {},
|
||||||
|
|
||||||
else => return error.Malformed,
|
else => return error.Malformed,
|
||||||
}
|
}
|
||||||
@@ -508,10 +508,10 @@ pub const Parser = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
fn isNameStart(b: u8) bool {
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
b == op.name_char_underscore or
|
b == opcode.name_char_underscore or
|
||||||
b == op.root_char or
|
b == opcode.root_char or
|
||||||
b == op.parent_prefix_char or
|
b == opcode.parent_prefix_char or
|
||||||
b == op.dual_name_prefix or
|
b == opcode.dual_name_prefix or
|
||||||
b == op.multi_name_prefix;
|
b == opcode.multi_name_prefix;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
//! Discovery backends (ACPI today, device-tree later) translate their native
|
//! Discovery backends (ACPI today, device-tree later) translate their native
|
||||||
//! hardware description into this one shape, so the rest of the kernel walks a
|
//! hardware description into this one shape, so the rest of the kernel walks a
|
||||||
//! plain `Device` tree without knowing which firmware described the machine —
|
//! plain `Device` tree without knowing which firmware described the machine —
|
||||||
//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `arch`
|
//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `architecture`
|
||||||
//! applies to the CPU.
|
//! applies to the CPU.
|
||||||
//!
|
//!
|
||||||
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
||||||
@@ -14,14 +14,14 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
|
||||||
/// The hardware primitives a discovery backend needs but can't express portably.
|
/// The hardware primitives a discovery backend needs but can't express portably.
|
||||||
/// The kernel injects an implementation (the arch VMM + port I/O), so the device
|
/// The kernel injects an implementation (the architecture VMM + port I/O), so the device
|
||||||
/// layer touches hardware without importing `arch` — the same discipline that lets
|
/// layer touches hardware without importing `architecture` — the same discipline that lets
|
||||||
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
||||||
pub const Hal = struct {
|
pub const Hal = struct {
|
||||||
/// Map a physical MMIO range and return the virtual address to reach it at.
|
/// Map a physical MMIO range and return the virtual address to reach it at.
|
||||||
/// The device layer dereferences the returned address and never learns how
|
/// The device layer dereferences the returned address and never learns how
|
||||||
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
||||||
mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64,
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
};
|
};
|
||||||
@@ -74,29 +74,29 @@ pub const Ids = struct {
|
|||||||
pci_device: ?u16 = null,
|
pci_device: ?u16 = null,
|
||||||
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
||||||
pci_class: ?u24 = null,
|
pci_class: ?u24 = null,
|
||||||
/// PCI bus/device/function packed as (bus << 8) | (dev << 3) | func — the key
|
/// PCI bus/device/function packed as (bus << 8) | (device << 3) | function — the key
|
||||||
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
||||||
pci_bdf: ?u16 = null,
|
pci_bdf: ?u16 = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
||||||
/// the busy case). Stored inline so a device is a single allocation.
|
/// the busy case). Stored inline so a device is a single allocation.
|
||||||
pub const max_resources = 8;
|
pub const maximum_resources = 8;
|
||||||
|
|
||||||
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
||||||
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
||||||
/// no per-node dynamic arrays to manage.
|
/// no per-node dynamic arrays to manage.
|
||||||
pub const Device = struct {
|
pub const Device = struct {
|
||||||
name_buf: [24]u8 = undefined,
|
name_buffer: [24]u8 = undefined,
|
||||||
name_len: u8 = 0,
|
name_len: u8 = 0,
|
||||||
class: DeviceClass = .unknown,
|
class: DeviceClass = .unknown,
|
||||||
ids: Ids = .{},
|
ids: Ids = .{},
|
||||||
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
||||||
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
||||||
/// how a backend encoded it.
|
/// how a backend encoded it.
|
||||||
hid_buf: [8]u8 = undefined,
|
hid_buffer: [8]u8 = undefined,
|
||||||
hid_len: u8 = 0,
|
hid_len: u8 = 0,
|
||||||
resources: [max_resources]Resource = undefined,
|
resources: [maximum_resources]Resource = undefined,
|
||||||
resource_count: u8 = 0,
|
resource_count: u8 = 0,
|
||||||
|
|
||||||
parent: ?*Device = null,
|
parent: ?*Device = null,
|
||||||
@@ -106,30 +106,30 @@ pub const Device = struct {
|
|||||||
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
||||||
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
||||||
pub fn name(self: *const Device) []const u8 {
|
pub fn name(self: *const Device) []const u8 {
|
||||||
return self.name_buf[0..self.name_len];
|
return self.name_buffer[0..self.name_len];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn setName(self: *Device, s: []const u8) void {
|
fn setName(self: *Device, s: []const u8) void {
|
||||||
const n: u8 = @intCast(@min(s.len, self.name_buf.len));
|
const n: u8 = @intCast(@min(s.len, self.name_buffer.len));
|
||||||
@memcpy(self.name_buf[0..n], s[0..n]);
|
@memcpy(self.name_buffer[0..n], s[0..n]);
|
||||||
self.name_len = n;
|
self.name_len = n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The device's hardware id string, or empty if none is set.
|
/// The device's hardware id string, or empty if none is set.
|
||||||
pub fn hid(self: *const Device) []const u8 {
|
pub fn hid(self: *const Device) []const u8 {
|
||||||
return self.hid_buf[0..self.hid_len];
|
return self.hid_buffer[0..self.hid_len];
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn setHid(self: *Device, s: []const u8) void {
|
pub fn setHid(self: *Device, s: []const u8) void {
|
||||||
const n: u8 = @intCast(@min(s.len, self.hid_buf.len));
|
const n: u8 = @intCast(@min(s.len, self.hid_buffer.len));
|
||||||
@memcpy(self.hid_buf[0..n], s[0..n]);
|
@memcpy(self.hid_buffer[0..n], s[0..n]);
|
||||||
self.hid_len = n;
|
self.hid_len = n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Record a resource. Silently drops beyond `max_resources` — discovery logs
|
/// Record a resource. Silently drops beyond `maximum_resources` — discovery logs
|
||||||
/// the truncation rather than failing the whole tree.
|
/// the truncation rather than failing the whole tree.
|
||||||
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
||||||
if (self.resource_count >= max_resources) return false;
|
if (self.resource_count >= maximum_resources) return false;
|
||||||
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
||||||
self.resource_count += 1;
|
self.resource_count += 1;
|
||||||
return true;
|
return true;
|
||||||
@@ -170,17 +170,17 @@ pub const DeviceTree = struct {
|
|||||||
self: *DeviceTree,
|
self: *DeviceTree,
|
||||||
parent: *Device,
|
parent: *Device,
|
||||||
class: DeviceClass,
|
class: DeviceClass,
|
||||||
dev_name: []const u8,
|
device_name: []const u8,
|
||||||
) !*Device {
|
) !*Device {
|
||||||
const d = try self.allocator.create(Device);
|
const d = try self.allocator.create(Device);
|
||||||
d.* = .{ .class = class, .parent = parent };
|
d.* = .{ .class = class, .parent = parent };
|
||||||
d.setName(dev_name);
|
d.setName(device_name);
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = d;
|
parent.first_child = d;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = d;
|
current.next_sibling = d;
|
||||||
}
|
}
|
||||||
return d;
|
return d;
|
||||||
}
|
}
|
||||||
@@ -202,18 +202,18 @@ fn firstOfClassIn(node: *Device, class: DeviceClass) ?*Device {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
fn dumpNode(device: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
||||||
const indent = @min(depth * 2, 40);
|
const indent = @min(depth * 2, 40);
|
||||||
|
|
||||||
var buf: [200]u8 = undefined;
|
var buffer: [200]u8 = undefined;
|
||||||
@memset(buf[0..indent], ' ');
|
@memset(buffer[0..indent], ' ');
|
||||||
const body = if (dev.hid_len != 0)
|
const body = if (device.hid_len != 0)
|
||||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}] hid={s}\n", .{ dev.name(), @tagName(dev.class), dev.hid() }) catch return
|
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s}\n", .{ device.name(), @tagName(device.class), device.hid() }) catch return
|
||||||
else
|
else
|
||||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}]\n", .{ dev.name(), @tagName(dev.class) }) catch return;
|
std.fmt.bufPrint(buffer[indent..], "{s} [{s}]\n", .{ device.name(), @tagName(device.class) }) catch return;
|
||||||
emit(buf[0 .. indent + body.len]);
|
emit(buffer[0 .. indent + body.len]);
|
||||||
|
|
||||||
for (dev.resources[0..dev.resource_count]) |r| {
|
for (device.resources[0..device.resource_count]) |r| {
|
||||||
var rbuf: [200]u8 = undefined;
|
var rbuf: [200]u8 = undefined;
|
||||||
const pad = @min(indent + 2, 42);
|
const pad = @min(indent + 2, 42);
|
||||||
@memset(rbuf[0..pad], ' ');
|
@memset(rbuf[0..pad], ' ');
|
||||||
@@ -225,6 +225,6 @@ fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void)
|
|||||||
emit(rbuf[0 .. pad + rline.len]);
|
emit(rbuf[0 .. pad + rline.len]);
|
||||||
}
|
}
|
||||||
|
|
||||||
var child = dev.first_child;
|
var child = device.first_child;
|
||||||
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
||||||
}
|
}
|
||||||
@@ -7,10 +7,10 @@
|
|||||||
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
||||||
//! parser in later is a local change here, not an architectural one.
|
//! parser in later is a local change here, not an architectural one.
|
||||||
|
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
|
|
||||||
/// Populate `dt` from a device-tree blob. Not implemented yet.
|
/// Populate `device_tree` from a device-tree blob. Not implemented yet.
|
||||||
pub fn discover(dt: *device.DeviceTree) !void {
|
pub fn discover(device_tree: *device_model.DeviceTree) !void {
|
||||||
_ = dt;
|
_ = device_tree;
|
||||||
return error.Unsupported;
|
return error.Unsupported;
|
||||||
}
|
}
|
||||||
+28
-28
@@ -1,42 +1,42 @@
|
|||||||
//! The firmware-agnostic discovery facade.
|
//! The firmware-agnostic discovery facade.
|
||||||
//!
|
//!
|
||||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
||||||
//! without ever naming ACPI or device-tree — the same way it imports `arch`
|
//! without ever naming ACPI or device-tree — the same way it imports `architecture`
|
||||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
||||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
||||||
//! because a single image — a future ARM kernel especially — may boot under
|
//! because a single image — a future ARM kernel especially — may boot under
|
||||||
//! either firmware. That's a deliberate divergence from `arch`, which is a
|
//! either firmware. That's a deliberate divergence from `architecture`, which is a
|
||||||
//! compile-time choice.
|
//! compile-time choice.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const acpi = @import("acpi.zig");
|
const acpi = @import("acpi.zig");
|
||||||
const power = @import("power.zig");
|
const power = @import("power.zig");
|
||||||
const devicetree = @import("devicetree.zig");
|
const devicetree = @import("device-tree.zig");
|
||||||
|
|
||||||
pub const DeviceTree = device.DeviceTree;
|
pub const DeviceTree = device_model.DeviceTree;
|
||||||
pub const Device = device.Device;
|
pub const Device = device_model.Device;
|
||||||
pub const DeviceClass = device.DeviceClass;
|
pub const DeviceClass = device_model.DeviceClass;
|
||||||
pub const Resource = device.Resource;
|
pub const Resource = device_model.Resource;
|
||||||
pub const ResourceKind = device.ResourceKind;
|
pub const ResourceKind = device_model.ResourceKind;
|
||||||
pub const Hal = device.Hal;
|
pub const Hal = device_model.Hal;
|
||||||
pub const PowerInfo = acpi.PowerInfo;
|
pub const PowerInformation = acpi.PowerInformation;
|
||||||
pub const AmlStats = acpi.AmlStats;
|
pub const AmlStats = acpi.AmlStats;
|
||||||
pub const PlatformInfo = acpi.PlatformInfo;
|
pub const PlatformInformation = acpi.PlatformInformation;
|
||||||
pub const RegAccess = acpi.RegAccess;
|
pub const RegisterAccess = acpi.RegisterAccess;
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
pub const IsoEntry = acpi.IsoEntry;
|
||||||
pub const Cpu = acpi.Cpu;
|
pub const Cpu = acpi.Cpu;
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||||
pub fn powerInfo() PowerInfo {
|
pub fn powerInformation() PowerInformation {
|
||||||
return acpi.power_info;
|
return acpi.power_information;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The scalar firmware facts the arch layer needs to avoid legacy assumptions
|
/// The scalar firmware facts the architecture layer needs to avoid legacy assumptions
|
||||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
||||||
pub fn platformInfo() PlatformInfo {
|
pub fn platformInformation() PlatformInformation {
|
||||||
return acpi.platform_info;
|
return acpi.platform_information;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||||
@@ -51,36 +51,36 @@ pub fn amlStats() AmlStats {
|
|||||||
/// bootstrap processor is actually running, so starting the rest is the pending SMP
|
/// bootstrap processor is actually running, so starting the rest is the pending SMP
|
||||||
/// step (see docs/smp.md). Borrowed from static storage populated by `discover`.
|
/// step (see docs/smp.md). Borrowed from static storage populated by `discover`.
|
||||||
pub fn cpus() []const Cpu {
|
pub fn cpus() []const Cpu {
|
||||||
return acpi.cpu_info.cpus[0..acpi.cpu_info.count];
|
return acpi.cpu_information.cpus[0..acpi.cpu_information.count];
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Non-zero only if enumeration found more processors than the static pool holds
|
/// Non-zero only if enumeration found more processors than the static pool holds
|
||||||
/// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent.
|
/// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent.
|
||||||
pub fn cpusDropped() usize {
|
pub fn cpusDropped() usize {
|
||||||
return acpi.cpu_info.dropped;
|
return acpi.cpu_information.dropped;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
||||||
/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for
|
/// primitives the backend needs (MMIO mapping for PCIe configuration space, port I/O for
|
||||||
/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up
|
/// ACPI registers); pass the architecture implementation. Errors leave nothing to clean up
|
||||||
/// beyond the tree's own allocations.
|
/// beyond the tree's own allocations.
|
||||||
pub fn discover(
|
pub fn discover(
|
||||||
boot_info: *const danos.BootInfo,
|
boot_information: *const danos.BootInformation,
|
||||||
allocator: std.mem.Allocator,
|
allocator: std.mem.Allocator,
|
||||||
hal: Hal,
|
hal: Hal,
|
||||||
) !DeviceTree {
|
) !DeviceTree {
|
||||||
var dt = try DeviceTree.init(allocator);
|
var device_tree = try DeviceTree.init(allocator);
|
||||||
|
|
||||||
if (boot_info.acpi_rsdp != 0) {
|
if (boot_information.acpi_rsdp != 0) {
|
||||||
try acpi.discover(boot_info.acpi_rsdp, &dt, hal);
|
try acpi.discover(boot_information.acpi_rsdp, &device_tree, hal);
|
||||||
} else {
|
} else {
|
||||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||||
// path is a stub, so this reports the machine described itself no way we
|
// path is a stub, so this reports the machine described itself no way we
|
||||||
// understand yet.
|
// understand yet.
|
||||||
try devicetree.discover(&dt);
|
try devicetree.discover(&device_tree);
|
||||||
}
|
}
|
||||||
|
|
||||||
return dt;
|
return device_tree;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||||
|
|||||||
+20
-20
@@ -9,8 +9,8 @@
|
|||||||
//! re-initialisation, a milestone of its own.
|
//! re-initialisation, a milestone of its own.
|
||||||
|
|
||||||
const acpi = @import("acpi.zig");
|
const acpi = @import("acpi.zig");
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const Hal = device.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||||
@@ -19,27 +19,27 @@ const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is acti
|
|||||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||||
pub fn enable(hal: Hal) void {
|
pub fn enable(hal: Hal) void {
|
||||||
const pi = acpi.power_info;
|
const pi = acpi.power_information;
|
||||||
if (!pi.pm1a_cnt.present()) return;
|
if (!pi.pm1a_cnt.present()) return;
|
||||||
if (readReg(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||||
|
|
||||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||||
var spins: usize = 0;
|
var spins: usize = 0;
|
||||||
while (readReg(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
const pi = acpi.power_info;
|
const pi = acpi.power_information;
|
||||||
|
|
||||||
// 1. The FADT reset register, when the firmware advertises support.
|
// 1. The FADT reset register, when the firmware advertises support.
|
||||||
if (pi.reset_supported and pi.reset.present()) {
|
if (pi.reset_supported and pi.reset.present()) {
|
||||||
writeReg(hal, pi.reset, pi.reset_value);
|
writeRegister(hal, pi.reset, pi.reset_value);
|
||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYS_RST).
|
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYSTEM_RST).
|
||||||
hal.pioWrite(1, 0xCF9, 0x0E);
|
hal.pioWrite(1, 0xCF9, 0x0E);
|
||||||
hal.pioWrite(1, 0xCF9, 0x06);
|
hal.pioWrite(1, 0xCF9, 0x06);
|
||||||
delay();
|
delay();
|
||||||
@@ -52,14 +52,14 @@ pub fn reboot(hal: Hal) void {
|
|||||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||||
pub fn shutdown(hal: Hal) void {
|
pub fn shutdown(hal: Hal) void {
|
||||||
enable(hal);
|
enable(hal);
|
||||||
const pi = acpi.power_info;
|
const pi = acpi.power_information;
|
||||||
const s5 = pi.s5 orelse return;
|
const s5 = pi.s5 orelse return;
|
||||||
|
|
||||||
if (pi.pm1a_cnt.present()) {
|
if (pi.pm1a_cnt.present()) {
|
||||||
writeReg(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||||
}
|
}
|
||||||
if (pi.pm1b_cnt.present()) {
|
if (pi.pm1b_cnt.present()) {
|
||||||
writeReg(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||||
}
|
}
|
||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
@@ -76,25 +76,25 @@ fn sleepValue(slp_typ: u8) u32 {
|
|||||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||||
if (reg.mmio) {
|
if (register.mmio) {
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true));
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
return p.*;
|
return p.*;
|
||||||
}
|
}
|
||||||
return hal.pioRead(reg.width, @intCast(reg.address));
|
return hal.pioRead(register.width, @intCast(register.address));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void {
|
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||||
if (reg.mmio) {
|
if (register.mmio) {
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true));
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
p.* = value;
|
p.* = value;
|
||||||
} else {
|
} else {
|
||||||
hal.pioWrite(reg.width, @intCast(reg.address), value);
|
hal.pioWrite(register.width, @intCast(register.address), value);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||||
/// the next method. The empty asm is an arch-neutral barrier that keeps the loop
|
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||||
/// from being optimised away.
|
/// from being optimised away.
|
||||||
fn delay() void {
|
fn delay() void {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
|
|||||||
@@ -18,17 +18,17 @@ pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool };
|
|||||||
|
|
||||||
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
|
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
|
||||||
// the legacy-safe assumptions so the code still works if discovery never ran.
|
// the legacy-safe assumptions so the code still works if discovery never ran.
|
||||||
var cfg_pic_present: bool = true;
|
var configuration_pic_present: bool = true;
|
||||||
var cfg_hpet_base: u64 = 0; // 0 = no HPET discovered
|
var configuration_hpet_base: u64 = 0; // 0 = no HPET discovered
|
||||||
var cfg_pm_timer: ?PmTimer = null;
|
var configuration_pm_timer: ?PmTimer = null;
|
||||||
/// Which reference the last calibration used, for logging.
|
/// Which reference the last calibration used, for logging.
|
||||||
var cal_source: []const u8 = "none";
|
var cal_source: []const u8 = "none";
|
||||||
|
|
||||||
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
|
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
|
||||||
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
|
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
|
||||||
cfg_pic_present = pic_present;
|
configuration_pic_present = pic_present;
|
||||||
cfg_hpet_base = hpet_base;
|
configuration_hpet_base = hpet_base;
|
||||||
cfg_pm_timer = pm_timer;
|
configuration_pm_timer = pm_timer;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
|
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
|
||||||
@@ -43,14 +43,15 @@ pub const timer_vector = 32;
|
|||||||
const spurious_vector = 47;
|
const spurious_vector = 47;
|
||||||
|
|
||||||
// LAPIC register offsets.
|
// LAPIC register offsets.
|
||||||
const reg_spurious = 0x0F0;
|
const register_spurious = 0x0F0;
|
||||||
const reg_eoi = 0x0B0;
|
const register_eoi = 0x0B0;
|
||||||
const reg_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
|
const register_id = 0x020; // this core's LAPIC id, in bits 24-31
|
||||||
const reg_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
|
const register_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
|
||||||
const reg_lvt_timer = 0x320;
|
const register_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
|
||||||
const reg_timer_initial = 0x380;
|
const register_lvt_timer = 0x320;
|
||||||
const reg_timer_current = 0x390;
|
const register_timer_initial = 0x380;
|
||||||
const reg_timer_divide = 0x3E0;
|
const register_timer_current = 0x390;
|
||||||
|
const register_timer_divide = 0x3E0;
|
||||||
|
|
||||||
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
|
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
|
||||||
|
|
||||||
@@ -90,11 +91,11 @@ fn rdtsc() u64 {
|
|||||||
return (@as(u64, high) << 32) | low;
|
return (@as(u64, high) << 32) | low;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn read(reg: u32) u32 {
|
fn read(register: u32) u32 {
|
||||||
return @as(*volatile u32, @ptrFromInt(base + reg)).*;
|
return @as(*volatile u32, @ptrFromInt(base + register)).*;
|
||||||
}
|
}
|
||||||
fn write(reg: u32, value: u32) void {
|
fn write(register: u32, value: u32) void {
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg)).* = value;
|
@as(*volatile u32, @ptrFromInt(base + register)).* = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
||||||
@@ -116,14 +117,14 @@ fn remapAndMaskPic() void {
|
|||||||
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
|
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
|
||||||
/// software-enable the APIC via its spurious-vector register.
|
/// software-enable the APIC via its spurious-vector register.
|
||||||
pub fn init() void {
|
pub fn init() void {
|
||||||
if (cfg_pic_present) remapAndMaskPic();
|
if (configuration_pic_present) remapAndMaskPic();
|
||||||
|
|
||||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||||
// Reach the LAPIC through the physmap (paging.init maps its page there).
|
// Reach the LAPIC through the physmap (paging.init maps its page there).
|
||||||
base = @intCast(danos.physToVirt(msr & 0xFFFFF000)); // physical base is bits 12+
|
base = @intCast(danos.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+
|
||||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||||
|
|
||||||
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
write(register_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Software-enable *this* core's Local APIC — the application-processor counterpart
|
/// Software-enable *this* core's Local APIC — the application-processor counterpart
|
||||||
@@ -133,7 +134,7 @@ pub fn init() void {
|
|||||||
pub fn initSecondary() void {
|
pub fn initSecondary() void {
|
||||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||||
write(reg_spurious, 0x100 | spurious_vector); // software enable
|
write(register_spurious, 0x100 | spurious_vector); // software enable
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
|
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
|
||||||
@@ -141,8 +142,8 @@ pub fn initSecondary() void {
|
|||||||
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
|
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
|
||||||
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
|
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
|
||||||
pub fn sendInit(apic_id: u32) void {
|
pub fn sendInit(apic_id: u32) void {
|
||||||
write(reg_icr_high, apic_id << 24);
|
write(register_icr_high, apic_id << 24);
|
||||||
write(reg_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
|
write(register_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
|
||||||
waitIcrIdle();
|
waitIcrIdle();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -150,13 +151,13 @@ pub fn sendInit(apic_id: u32) void {
|
|||||||
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
|
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
|
||||||
/// sent twice after the INIT; both calls block until delivery completes.
|
/// sent twice after the INIT; both calls block until delivery completes.
|
||||||
pub fn sendStartup(apic_id: u32, vector: u8) void {
|
pub fn sendStartup(apic_id: u32, vector: u8) void {
|
||||||
write(reg_icr_high, apic_id << 24);
|
write(register_icr_high, apic_id << 24);
|
||||||
write(reg_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
|
write(register_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
|
||||||
waitIcrIdle();
|
waitIcrIdle();
|
||||||
}
|
}
|
||||||
|
|
||||||
fn waitIcrIdle() void {
|
fn waitIcrIdle() void {
|
||||||
while (read(reg_icr_low) & icr_delivery_pending != 0) {}
|
while (read(register_icr_low) & icr_delivery_pending != 0) {}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The calibration window: we time everything against a 10 ms reference interval.
|
/// The calibration window: we time everything against a 10 ms reference interval.
|
||||||
@@ -180,7 +181,7 @@ pub fn calibrate() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// 2. The discovered HPET.
|
// 2. The discovered HPET.
|
||||||
if (!done and cfg_hpet_base != 0) {
|
if (!done and configuration_hpet_base != 0) {
|
||||||
if (hpetHz()) |hpet_hz| {
|
if (hpetHz()) |hpet_hz| {
|
||||||
measure(hpet_hz, hpetMask(), readHpet);
|
measure(hpet_hz, hpetMask(), readHpet);
|
||||||
cal_source = "hpet";
|
cal_source = "hpet";
|
||||||
@@ -190,7 +191,7 @@ pub fn calibrate() void {
|
|||||||
|
|
||||||
// 3. The ACPI PM timer (fixed 3.579545 MHz).
|
// 3. The ACPI PM timer (fixed 3.579545 MHz).
|
||||||
if (!done) {
|
if (!done) {
|
||||||
if (cfg_pm_timer) |pt| {
|
if (configuration_pm_timer) |pt| {
|
||||||
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
|
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
|
||||||
cal_source = "pm-timer";
|
cal_source = "pm-timer";
|
||||||
done = true;
|
done = true;
|
||||||
@@ -212,23 +213,23 @@ pub fn calibrate() void {
|
|||||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Run the LAPIC timer one-shot from its max count while a monotonic reference
|
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||||
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
|
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
|
||||||
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
|
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
|
||||||
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
|
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
|
||||||
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
|
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
|
||||||
|
|
||||||
write(reg_timer_divide, timer_divide_16);
|
write(register_timer_divide, timer_divide_16);
|
||||||
write(reg_lvt_timer, lvt_masked);
|
write(register_lvt_timer, lvt_masked);
|
||||||
write(reg_timer_initial, 0xFFFFFFFF);
|
write(register_timer_initial, 0xFFFFFFFF);
|
||||||
|
|
||||||
const ref0 = refNow();
|
const ref0 = refNow();
|
||||||
const tsc0 = rdtsc();
|
const tsc0 = rdtsc();
|
||||||
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
|
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
|
||||||
const tsc1 = rdtsc();
|
const tsc1 = rdtsc();
|
||||||
|
|
||||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
const elapsed = 0xFFFFFFFF - read(register_timer_current);
|
||||||
write(reg_timer_initial, 0);
|
write(register_timer_initial, 0);
|
||||||
|
|
||||||
ticks_per_ms = elapsed / calib_ms;
|
ticks_per_ms = elapsed / calib_ms;
|
||||||
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
|
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
|
||||||
@@ -240,9 +241,9 @@ fn calibratePit() void {
|
|||||||
const pit_hz = 1_193_182;
|
const pit_hz = 1_193_182;
|
||||||
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
||||||
|
|
||||||
write(reg_timer_divide, timer_divide_16);
|
write(register_timer_divide, timer_divide_16);
|
||||||
write(reg_lvt_timer, lvt_masked);
|
write(register_lvt_timer, lvt_masked);
|
||||||
write(reg_timer_initial, 0xFFFFFFFF);
|
write(register_timer_initial, 0xFFFFFFFF);
|
||||||
|
|
||||||
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
||||||
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
||||||
@@ -255,8 +256,8 @@ fn calibratePit() void {
|
|||||||
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
|
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
|
||||||
const tsc_end = rdtsc();
|
const tsc_end = rdtsc();
|
||||||
|
|
||||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
const elapsed = 0xFFFFFFFF - read(register_timer_current);
|
||||||
write(reg_timer_initial, 0);
|
write(register_timer_initial, 0);
|
||||||
|
|
||||||
ticks_per_ms = elapsed / calib_ms;
|
ticks_per_ms = elapsed / calib_ms;
|
||||||
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
|
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
|
||||||
@@ -292,19 +293,19 @@ fn cpuid(leaf: u32) CpuidRegs {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
|
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
|
||||||
// 64-bit-counter capable), general config at +0x10, main counter at +0xF0.
|
// 64-bit-counter capable), general configuration at +0x10, main counter at +0xF0.
|
||||||
fn hpetRead64(off: usize) u64 {
|
fn hpetRead64(off: usize) u64 {
|
||||||
return @as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).*;
|
return @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).*;
|
||||||
}
|
}
|
||||||
fn hpetWrite64(off: usize, value: u64) void {
|
fn hpetWrite64(off: usize, value: u64) void {
|
||||||
@as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).* = value;
|
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||||
/// Maps the HPET into the physmap and switches cfg_hpet_base to that virtual
|
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||||
/// address, so the register accessors reach it without the identity map.
|
/// address, so the register accessors reach it without the identity map.
|
||||||
fn hpetHz() ?u64 {
|
fn hpetHz() ?u64 {
|
||||||
cfg_hpet_base = paging.mapMmio(cfg_hpet_base, 0x400, true);
|
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||||
const caps = hpetRead64(0x00);
|
const caps = hpetRead64(0x00);
|
||||||
const period_fs = caps >> 32; // femtoseconds per tick
|
const period_fs = caps >> 32; // femtoseconds per tick
|
||||||
if (period_fs == 0) return null;
|
if (period_fs == 0) return null;
|
||||||
@@ -322,7 +323,7 @@ fn readHpet() u64 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn readPmTimer() u64 {
|
fn readPmTimer() u64 {
|
||||||
const pt = cfg_pm_timer.?;
|
const pt = configuration_pm_timer.?;
|
||||||
// MMIO PM timer via the physmap (mapMmio is idempotent); the common case is
|
// MMIO PM timer via the physmap (mapMmio is idempotent); the common case is
|
||||||
// a legacy I/O port.
|
// a legacy I/O port.
|
||||||
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*;
|
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*;
|
||||||
@@ -334,9 +335,9 @@ fn readPmTimer() u64 {
|
|||||||
pub fn initTimer(hz: u32) void {
|
pub fn initTimer(hz: u32) void {
|
||||||
timer_hz = hz;
|
timer_hz = hz;
|
||||||
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
||||||
write(reg_timer_divide, timer_divide_16);
|
write(register_timer_divide, timer_divide_16);
|
||||||
write(reg_lvt_timer, timer_vector | lvt_periodic);
|
write(register_lvt_timer, timer_vector | lvt_periodic);
|
||||||
write(reg_timer_initial, @intCast(count));
|
write(register_timer_initial, @intCast(count));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Configured periodic-interrupt frequency (Hz).
|
/// Configured periodic-interrupt frequency (Hz).
|
||||||
@@ -376,7 +377,7 @@ pub fn millis() u64 {
|
|||||||
|
|
||||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||||
pub fn eoi() void {
|
pub fn eoi() void {
|
||||||
write(reg_eoi, 0);
|
write(register_eoi, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
/// Optional callback run each tick (the scheduler registers it for preemption).
|
||||||
@@ -390,10 +391,21 @@ pub fn setTickHook(hook: *const fn () void) void {
|
|||||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
||||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
||||||
pub fn timerTick() void {
|
pub fn timerTick() void {
|
||||||
|
// Acknowledge before the tick hook: `on_tick` is the scheduler, which may switch
|
||||||
|
// tasks and not return promptly, and the LAPIC mustn't wait on it to deliver the
|
||||||
|
// next interrupt. (Each device handler now owns its own EOI — see
|
||||||
|
// `idt.interruptDispatch` — because a *routed* interrupt must be masked at the
|
||||||
|
// I/O APIC before it is acknowledged, an ordering the dispatcher can't impose.)
|
||||||
|
eoi();
|
||||||
tick_count +%= 1;
|
tick_count +%= 1;
|
||||||
if (on_tick) |hook| hook();
|
if (on_tick) |hook| hook();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// This core's Local APIC id — the interrupt destination for `routeGsi`.
|
||||||
|
pub fn localId() u8 {
|
||||||
|
return @truncate(read(register_id) >> 24);
|
||||||
|
}
|
||||||
|
|
||||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
/// Number of timer ticks so far. Volatile load: the count is bumped
|
||||||
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
||||||
pub fn ticks() u64 {
|
pub fn ticks() u64 {
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
//! x86_64 CPU operations. This is the "architecture" module: the generic kernel imports
|
||||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
//! it as `@import("architecture")` and never names x86_64 directly, so a second
|
||||||
//! architecture is added by pointing that module at a different directory in
|
//! architecture is added by pointing that module at a different directory in
|
||||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||||
|
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
const gdt = @import("gdt.zig");
|
const gdt = @import("gdt.zig");
|
||||||
const tss = @import("tss.zig");
|
const tss = @import("tss.zig");
|
||||||
const idt = @import("idt.zig");
|
const idt = @import("idt.zig");
|
||||||
@@ -15,7 +15,7 @@ const apic = @import("apic.zig");
|
|||||||
const ioapic = @import("ioapic.zig");
|
const ioapic = @import("ioapic.zig");
|
||||||
const io = @import("io.zig");
|
const io = @import("io.zig");
|
||||||
const smp = @import("smp.zig");
|
const smp = @import("smp.zig");
|
||||||
const pcpu = @import("percpu.zig");
|
const pcpu = @import("per-cpu.zig");
|
||||||
|
|
||||||
/// The saved register/trap frame passed to a fault handler.
|
/// The saved register/trap frame passed to a fault handler.
|
||||||
pub const CpuState = idt.CpuState;
|
pub const CpuState = idt.CpuState;
|
||||||
@@ -50,18 +50,18 @@ pub fn faultAddress(state: *const CpuState) ?u64 {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- syscall ABI ------------------------------------------------------------
|
// --- system_call ABI ------------------------------------------------------------
|
||||||
// The System V-style register convention (number in rax, arguments in
|
// The System V-style register convention (number in rax, arguments in
|
||||||
// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic
|
// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic
|
||||||
// dispatcher never names a register.
|
// dispatcher never names a register.
|
||||||
|
|
||||||
/// The syscall number the user program passed.
|
/// The system_call number the user program passed.
|
||||||
pub fn syscallNumber(state: *const CpuState) u64 {
|
pub fn systemCallNumber(state: *const CpuState) u64 {
|
||||||
return state.rax;
|
return state.rax;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Positional syscall argument `n`.
|
/// Positional system_call argument `n`.
|
||||||
pub fn syscallArg(state: *const CpuState, n: u8) u64 {
|
pub fn systemCallArg(state: *const CpuState, n: u8) u64 {
|
||||||
return switch (n) {
|
return switch (n) {
|
||||||
0 => state.rdi,
|
0 => state.rdi,
|
||||||
1 => state.rsi,
|
1 => state.rsi,
|
||||||
@@ -73,17 +73,17 @@ pub fn syscallArg(state: *const CpuState, n: u8) u64 {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write the syscall's return value into the frame — the entry paths restore
|
/// Write the system_call's return value into the frame — the entry paths restore
|
||||||
/// user registers from it.
|
/// user registers from it.
|
||||||
pub fn setSyscallResult(state: *CpuState, value: u64) void {
|
pub fn setSystemCallResult(state: *CpuState, value: u64) void {
|
||||||
state.rax = value;
|
state.rax = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write a *second* syscall return value (rdx here — restored by both the
|
/// Write a *second* system_call return value (rdx here — restored by both the
|
||||||
/// syscall/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by
|
/// system_call/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by
|
||||||
/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the
|
/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the
|
||||||
/// message length in rax.
|
/// message length in rax.
|
||||||
pub fn setSyscallResult2(state: *CpuState, value: u64) void {
|
pub fn setSystemCallResult2(state: *CpuState, value: u64) void {
|
||||||
state.rdx = value;
|
state.rdx = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -126,14 +126,14 @@ pub fn init() void {
|
|||||||
gdt.init();
|
gdt.init();
|
||||||
tss.init();
|
tss.init();
|
||||||
idt.init();
|
idt.init();
|
||||||
pcpu.initSyscall();
|
pcpu.initSystemCall();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||||
/// kernel's segment layout). Call once the frame allocator is up.
|
/// kernel's segment layout). Call once the frame allocator is up.
|
||||||
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void {
|
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||||
paging.init(allocFrame, freeFrame, boot_info);
|
paging.init(allocFrame, freeFrame, boot_information);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a new address space (returns the physical address of its root table —
|
/// Create a new address space (returns the physical address of its root table —
|
||||||
@@ -150,26 +150,26 @@ pub fn destroyAddressSpace(root: u64) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Map a user page into address space `root` (W^X is the caller's contract).
|
/// Map a user page into address space `root` (W^X is the caller's contract).
|
||||||
pub fn mapUserPageInto(root: u64, virt: u64, phys: u64, writable: bool, executable: bool) void {
|
pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||||
paging.mapUserInto(root, virt, phys, writable, executable);
|
paging.mapUserInto(root, virtual, physical, writable, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
||||||
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
||||||
pub fn mapUserDeviceInto(root: u64, virt: u64, phys: u64, len: u64) void {
|
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||||
paging.mapUserDeviceInto(root, virt, phys, len);
|
paging.mapUserDeviceInto(root, virtual, physical, len);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||||
paging.map(virt, phys, writable);
|
paging.map(virtual, physical, writable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a device MMIO range and return the virtual address to reach it at. This
|
/// Map a device MMIO range and return the virtual address to reach it at. This
|
||||||
/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and
|
/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and
|
||||||
/// never exposes how the mapping is placed.
|
/// never exposes how the mapping is placed.
|
||||||
pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 {
|
pub fn mapMmio(physical: u64, len: u64, writable: bool) u64 {
|
||||||
return paging.mapMmio(phys, len, writable);
|
return paging.mapMmio(physical, len, writable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The kernel's page-table root (physical), shared into every address space.
|
/// The kernel's page-table root (physical), shared into every address space.
|
||||||
@@ -192,7 +192,7 @@ pub fn activePageTable() u64 {
|
|||||||
|
|
||||||
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
||||||
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
|
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
|
||||||
/// per-CPU `kernel_rsp` (for the syscall stub, which switches by hand). Updated
|
/// per-CPU `kernel_rsp` (for the system_call stub, which switches by hand). Updated
|
||||||
/// by the scheduler when it switches to a user task.
|
/// by the scheduler when it switches to a user task.
|
||||||
pub fn setKernelStack(cpu: usize, top: usize) void {
|
pub fn setKernelStack(cpu: usize, top: usize) void {
|
||||||
tss.rsp0Ptr(cpu).* = top;
|
tss.rsp0Ptr(cpu).* = top;
|
||||||
@@ -200,27 +200,27 @@ pub fn setKernelStack(cpu: usize, top: usize) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Remove a kernel mapping.
|
/// Remove a kernel mapping.
|
||||||
pub fn unmapPage(virt: u64) void {
|
pub fn unmapPage(virtual: u64) void {
|
||||||
paging.unmap(virt);
|
paging.unmap(virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Remove a page mapping from address space `root` (for munmap of user pages).
|
/// Remove a page mapping from address space `root` (for munmap of user pages).
|
||||||
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
|
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
|
||||||
pub fn unmapUserPageInto(root: u64, virt: u64) void {
|
pub fn unmapUserPageInto(root: u64, virtual: u64) void {
|
||||||
paging.unmapInto(root, virt);
|
paging.unmapInto(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve `virt` to its physical address in the address space rooted at `root`
|
/// Resolve `virtual` to its physical address in the address space rooted at `root`
|
||||||
/// (any address space, not just the live one), or null if unmapped. Used to find
|
/// (any address space, not just the live one), or null if unmapped. Used to find
|
||||||
/// the frame behind a user page for munmap, and for cross-address-space copies.
|
/// the frame behind a user page for munmap, and for cross-address-space copies.
|
||||||
pub fn translate(root: u64, virt: u64) ?u64 {
|
pub fn translate(root: u64, virtual: u64) ?u64 {
|
||||||
return paging.translateIn(root, virt);
|
return paging.translateIn(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||||
/// W^X: code read-only + executable, data writable + no-execute.
|
/// W^X: code read-only + executable, data writable + no-execute.
|
||||||
pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void {
|
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||||
paging.mapUser(virt, phys, writable, executable);
|
paging.mapUser(virtual, physical, writable, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- ring 3 entry/exit -----------------------------------------------------
|
// --- ring 3 entry/exit -----------------------------------------------------
|
||||||
@@ -233,27 +233,27 @@ pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void
|
|||||||
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
|
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
|
||||||
|
|
||||||
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
|
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
|
||||||
/// `enter_user` had returned (defined in isr.s). Called by the exit syscall.
|
/// `enter_user` had returned (defined in isr.s). Called by the exit system_call.
|
||||||
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
||||||
|
|
||||||
/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the
|
/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the
|
||||||
/// caller's CPU index; the arch layer can't ask the scheduler). Returns after the
|
/// caller's CPU index; the architecture layer can't ask the scheduler). Returns after the
|
||||||
/// user program exits via syscall. Interrupts are disabled on return (the exit
|
/// user program exits via system_call. Interrupts are disabled on return (the exit
|
||||||
/// arrives through an interrupt gate) — the caller re-enables.
|
/// arrives through an interrupt gate) — the caller re-enables.
|
||||||
pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void {
|
pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void {
|
||||||
enter_user(entry, stack_top, tss.rsp0Ptr(cpu));
|
enter_user(entry, stack_top, tss.rsp0Ptr(cpu));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Never returns to the user program: unwind to the kernel context that called
|
/// Never returns to the user program: unwind to the kernel context that called
|
||||||
/// `enterUser`. For the exit syscall's handler.
|
/// `enterUser`. For the exit system_call's handler.
|
||||||
pub fn userExit() noreturn {
|
pub fn userExit() noreturn {
|
||||||
user_exit_to_kernel();
|
user_exit_to_kernel();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Register the handler for the user syscall gate (int 0x80, vector 128). The
|
/// Register the handler for the user system_call gate (int 0x80, vector 128). The
|
||||||
/// handler may write the trap frame (see `setSyscallResult`).
|
/// handler may write the trap frame (see `setSystemCallResult`).
|
||||||
pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void {
|
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
|
||||||
idt.setSyscallHandler(handler);
|
idt.setSystemCallHandler(handler);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
||||||
@@ -266,7 +266,7 @@ pub fn setCpuLocal(cpu: usize, ptr: usize) void {
|
|||||||
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
|
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
|
||||||
/// core sees its own without locking. Valid in any ring-0 context.
|
/// core sees its own without locking. Valid in any ring-0 context.
|
||||||
pub fn cpuLocal() usize {
|
pub fn cpuLocal() usize {
|
||||||
return pcpu.sched();
|
return pcpu.scheduler();
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- SMP: application-processor bring-up ----------------------------------
|
// --- SMP: application-processor bring-up ----------------------------------
|
||||||
@@ -275,8 +275,8 @@ pub fn cpuLocal() usize {
|
|||||||
/// The frame stays inert (zeroed, non-executable) between wakes and is armed only
|
/// The frame stays inert (zeroed, non-executable) between wakes and is armed only
|
||||||
/// while a core is climbing — so a core can be (re)woken at any time (retry, or a
|
/// while a core is climbing — so a core can be (re)woken at any time (retry, or a
|
||||||
/// future power manager) without leaving an executable page resident. See smp.zig.
|
/// future power manager) without leaving an executable page resident. See smp.zig.
|
||||||
pub fn setTrampolinePage(phys: u64) void {
|
pub fn setTrampolinePage(physical: u64) void {
|
||||||
smp.setTrampolinePage(phys);
|
smp.setTrampolinePage(physical);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on
|
/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on
|
||||||
@@ -290,7 +290,7 @@ pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize)
|
|||||||
return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4());
|
return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4());
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Register the generic entry a woken AP jumps to once its arch state is up (its own
|
/// Register the generic entry a woken AP jumps to once its architecture state is up (its own
|
||||||
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
|
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
|
||||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||||
smp.setSecondaryEntry(entry);
|
smp.setSecondaryEntry(entry);
|
||||||
@@ -316,22 +316,22 @@ pub fn trampolinePage() u64 {
|
|||||||
return smp.trampolinePage();
|
return smp.trampolinePage();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether the page at `virt` is currently mapped executable (present, NX clear).
|
/// Whether the page at `virtual` is currently mapped executable (present, NX clear).
|
||||||
pub fn pageExecutable(virt: u64) bool {
|
pub fn pageExecutable(virtual: u64) bool {
|
||||||
return paging.isExecutable(virt);
|
return paging.isExecutable(virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Kernel tick rate (the scheduler's time quantum), from config.
|
/// Kernel tick rate (the scheduler's time quantum), from configuration.
|
||||||
pub const timer_hz = config.timer_hz;
|
pub const timer_hz = parameters.timer_hz;
|
||||||
|
|
||||||
/// The ACPI PM timer, as a calibration reference (re-exported for the config).
|
/// The ACPI PM timer, as a calibration reference (re-exported for the configuration).
|
||||||
pub const PmTimer = apic.PmTimer;
|
pub const PmTimer = apic.PmTimer;
|
||||||
/// A MADT interrupt-source override (re-exported for the config).
|
/// A MADT interrupt-source override (re-exported for the configuration).
|
||||||
pub const IsoEntry = ioapic.IsoEntry;
|
pub const IsoEntry = ioapic.IsoEntry;
|
||||||
|
|
||||||
/// Discovered platform facts the arch layer needs so it makes no legacy
|
/// Discovered platform facts the architecture layer needs so it makes no legacy
|
||||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
||||||
pub const PlatformConfig = struct {
|
pub const PlatformConfiguration = struct {
|
||||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
||||||
pic_present: bool = true,
|
pic_present: bool = true,
|
||||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
||||||
@@ -345,18 +345,18 @@ pub const PlatformConfig = struct {
|
|||||||
overrides: []const IsoEntry = &.{},
|
overrides: []const IsoEntry = &.{},
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Apply the discovered platform config. Must run before `startTimer` (the timer
|
/// Apply the discovered platform configuration. Must run before `startTimer` (the timer
|
||||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
||||||
/// Maps + masks the I/O APIC immediately.
|
/// Maps + masks the I/O APIC immediately.
|
||||||
pub fn configurePlatform(cfg: PlatformConfig) void {
|
pub fn configurePlatform(configuration: PlatformConfiguration) void {
|
||||||
apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer);
|
apic.configure(configuration.pic_present, configuration.hpet_base, configuration.pm_timer);
|
||||||
ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides);
|
ioapic.configure(configuration.ioapic_base, configuration.ioapic_gsi_base, configuration.overrides);
|
||||||
ioapic.init();
|
ioapic.init();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
||||||
pub fn serialReconfigure(is_mmio: bool, addr: u64) void {
|
pub fn serialReconfigure(is_mmio: bool, address: u64) void {
|
||||||
serial.reconfigure(is_mmio, addr);
|
serial.reconfigure(is_mmio, address);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
||||||
@@ -373,6 +373,43 @@ pub fn irqRouteRaw(n: u32) u32 {
|
|||||||
return ioapic.entryLow(n);
|
return ioapic.entryLow(n);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- device-IRQ plumbing, for src/kernel/irq.zig -----------------------------
|
||||||
|
//
|
||||||
|
// The generic IRQ layer speaks GSIs and vectors; everything below hides the fact
|
||||||
|
// that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The
|
||||||
|
// vector window is bounded by the stubs isr.s actually emits: `gate_count` = 48,
|
||||||
|
// vector 32 is the LAPIC timer and 47 is the spurious vector, leaving 33..46.
|
||||||
|
|
||||||
|
pub const irq_vector_base: u8 = 33;
|
||||||
|
pub const irq_vector_count: u8 = 14; // 33..46 inclusive
|
||||||
|
|
||||||
|
/// True if `gsi` is one this machine's interrupt router can deliver.
|
||||||
|
pub fn irqOwnsGsi(gsi: u32) bool {
|
||||||
|
return ioapic.ownsGsi(gsi);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Install `handler` on `vector` (an absolute IDT gate index).
|
||||||
|
pub fn irqSetHandler(vector: u8, handler: *const fn () void) void {
|
||||||
|
idt.setHandler(vector, handler);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Route `gsi` to `vector` on *this* core, masked. Unmask with `irqUnmask` once bound.
|
||||||
|
pub fn irqRoute(gsi: u32, vector: u8, level: bool, active_low: bool) void {
|
||||||
|
ioapic.routeGsi(gsi, vector, apic.localId(), level, active_low);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn irqMask(gsi: u32) void {
|
||||||
|
ioapic.maskGsi(gsi);
|
||||||
|
}
|
||||||
|
pub fn irqUnmask(gsi: u32) void {
|
||||||
|
ioapic.unmaskGsi(gsi);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Acknowledge the interrupt currently in service on this core's LAPIC.
|
||||||
|
pub fn irqEoi() void {
|
||||||
|
apic.eoi();
|
||||||
|
}
|
||||||
|
|
||||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
/// Enable the Local APIC, calibrate its timer against the best available reference
|
||||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
||||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
||||||
|
|||||||
@@ -9,7 +9,7 @@
|
|||||||
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
|
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
|
||||||
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
|
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
|
||||||
|
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
|
|
||||||
/// Selectors into the table (index * 8). Same on every core's GDT.
|
/// Selectors into the table (index * 8). Same on every core's GDT.
|
||||||
pub const kernel_code = 0x08;
|
pub const kernel_code = 0x08;
|
||||||
@@ -24,7 +24,7 @@ pub const tss_selector = 0x28;
|
|||||||
pub const user_code_rpl3 = user_code | 3;
|
pub const user_code_rpl3 = user_code | 3;
|
||||||
pub const user_data_rpl3 = user_data | 3;
|
pub const user_data_rpl3 = user_data | 3;
|
||||||
|
|
||||||
const max_cpus = config.max_cpus;
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
|
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
|
||||||
|
|
||||||
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
|
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
|
||||||
@@ -47,7 +47,7 @@ const template = [entries]u64{
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
|
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
|
||||||
var gdts = [_][entries]u64{template} ** max_cpus;
|
var gdts = [_][entries]u64{template} ** maximum_cpus;
|
||||||
|
|
||||||
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
|
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
|
||||||
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
|
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
|
||||||
|
|||||||
@@ -9,7 +9,6 @@
|
|||||||
|
|
||||||
const gdt = @import("gdt.zig");
|
const gdt = @import("gdt.zig");
|
||||||
const tss = @import("tss.zig");
|
const tss = @import("tss.zig");
|
||||||
const apic = @import("apic.zig");
|
|
||||||
|
|
||||||
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
|
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
|
||||||
/// range 32-47, which covers the timer and the spurious vector).
|
/// range 32-47, which covers the timer and the spurious vector).
|
||||||
@@ -26,17 +25,17 @@ pub fn setHandler(vector: usize, handler: Handler) void {
|
|||||||
handlers[vector] = handler;
|
handlers[vector] = handler;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The ring-3 syscall gate's vector (`int $0x80`, the classic choice — well away
|
/// The ring-3 system_call gate's vector (`int $0x80`, the classic choice — well away
|
||||||
/// from the device range) and its handler. Unlike device handlers, a syscall
|
/// from the device range) and its handler. Unlike device handlers, a system_call
|
||||||
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
|
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
|
||||||
/// user registers and writes rax as the return value, which isr_common then
|
/// user registers and writes rax as the return value, which isr_common then
|
||||||
/// restores into the user context.
|
/// restores into the user context.
|
||||||
pub const syscall_vector = 128;
|
pub const system_call_vector = 128;
|
||||||
|
|
||||||
var syscall_handler: ?*const fn (*CpuState) void = null;
|
var system_call_handler: ?*const fn (*CpuState) void = null;
|
||||||
|
|
||||||
pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void {
|
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
|
||||||
syscall_handler = handler;
|
system_call_handler = handler;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The register + trap frame the ISR stubs build on the stack, laid out so the
|
/// The register + trap frame the ISR stubs build on the stack, laid out so the
|
||||||
@@ -141,13 +140,13 @@ pub fn init() void {
|
|||||||
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
|
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
|
||||||
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
|
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
|
||||||
idt[8].ist = tss.double_fault_ist;
|
idt[8].ist = tss.double_fault_ist;
|
||||||
// The syscall gate. Installed outside the 0..gate_count loop (stubs 48-127
|
// The system_call gate. Installed outside the 0..gate_count loop (stubs 48-127
|
||||||
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
|
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
|
||||||
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
|
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
|
||||||
// the ring-3 exit path relies on.
|
// the ring-3 exit path relies on.
|
||||||
const syscall_stub = @extern(*const anyopaque, .{ .name = "isr128" });
|
const system_call_stub = @extern(*const anyopaque, .{ .name = "isr128" });
|
||||||
setGate(syscall_vector, @intFromPtr(syscall_stub));
|
setGate(system_call_vector, @intFromPtr(system_call_stub));
|
||||||
idt[syscall_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
|
idt[system_call_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
|
||||||
loadOnThisCpu();
|
loadOnThisCpu();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -168,15 +167,17 @@ pub fn loadOnThisCpu() void {
|
|||||||
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
||||||
if (state.vector < 32) {
|
if (state.vector < 32) {
|
||||||
on_fault(state); // CPU exception — never returns
|
on_fault(state); // CPU exception — never returns
|
||||||
} else if (state.vector == syscall_vector) {
|
} else if (state.vector == system_call_vector) {
|
||||||
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
|
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
|
||||||
if (syscall_handler) |handler| handler(state);
|
if (system_call_handler) |handler| handler(state);
|
||||||
} else if (handlers[state.vector]) |handler| {
|
} else if (handlers[state.vector]) |handler| {
|
||||||
// Acknowledge before running the handler: a handler that switches tasks
|
// The handler owns its EOI. It used to be issued here, before the call —
|
||||||
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
|
// correct for the LAPIC timer, but impossible to reconcile with a
|
||||||
// it to deliver the next interrupt. Fine for edge-triggered sources like
|
// level-triggered device line, which must be **masked at the I/O APIC
|
||||||
// the timer; a level-triggered device would need EOI after handling.
|
// before** it is acknowledged or it redelivers instantly and storms
|
||||||
apic.eoi();
|
// (the driver that would quiet it lives in ring 3 and hasn't run yet).
|
||||||
|
// Only the handler knows which discipline its source needs, so only the
|
||||||
|
// handler can sequence it. See apic.timerTick and irq.dispatch.
|
||||||
handler();
|
handler();
|
||||||
}
|
}
|
||||||
// else: spurious/unhandled device interrupt — don't acknowledge it
|
// else: spurious/unhandled device interrupt — don't acknowledge it
|
||||||
|
|||||||
@@ -2,11 +2,15 @@
|
|||||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
||||||
//! ACPI's MADT (via discovery), never assumed.
|
//! ACPI's MADT (via discovery), never assumed.
|
||||||
//!
|
//!
|
||||||
//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own
|
//! `init` maps the I/O APIC and **masks every input** — the correct quiescent state
|
||||||
//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now
|
//! on a legacy-free machine. Lines are then unmasked one at a time, as user-space
|
||||||
//! is `init`, which maps the I/O APIC and **masks every input**, the correct
|
//! drivers bind them (`routeGsi`/`unmaskGsi`, driven by src/kernel/irq.zig).
|
||||||
//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real
|
//!
|
||||||
//! device driver (a keyboard, say).
|
//! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ**
|
||||||
|
//! and resolves it through the MADT overrides — for in-kernel use, and still without
|
||||||
|
//! a caller. `routeGsi` takes a **GSI** directly, which is what a device's own
|
||||||
|
//! routing capability names (e.g. the HPET's `Tn_INT_ROUTE_CAP`), and is the path a
|
||||||
|
//! bound driver interrupt takes.
|
||||||
|
|
||||||
const paging = @import("paging.zig");
|
const paging = @import("paging.zig");
|
||||||
|
|
||||||
@@ -16,14 +20,14 @@ pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
|||||||
|
|
||||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
var base: u64 = 0; // 0 = no I/O APIC discovered
|
||||||
var gsi_base: u32 = 0;
|
var gsi_base: u32 = 0;
|
||||||
var max_entries: u32 = 0;
|
var maximum_entries: u32 = 0;
|
||||||
var overrides: [16]IsoEntry = undefined;
|
var overrides: [16]IsoEntry = undefined;
|
||||||
var override_count: usize = 0;
|
var override_count: usize = 0;
|
||||||
|
|
||||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
||||||
const reg_ioregsel = 0x00;
|
const register_ioregsel = 0x00;
|
||||||
const reg_iowin = 0x10;
|
const register_iowin = 0x10;
|
||||||
const reg_version = 0x01;
|
const register_version = 0x01;
|
||||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
||||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
const redir_mask = 1 << 16; // mask bit in the low dword
|
||||||
|
|
||||||
@@ -35,18 +39,18 @@ pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry)
|
|||||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn regRead(index: u32) u32 {
|
fn registerRead(index: u32) u32 {
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
|
||||||
return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*;
|
return @as(*volatile u32, @ptrFromInt(base + register_iowin)).*;
|
||||||
}
|
}
|
||||||
fn regWrite(index: u32, value: u32) void {
|
fn registerWrite(index: u32, value: u32) void {
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value;
|
@as(*volatile u32, @ptrFromInt(base + register_iowin)).* = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
fn writeEntry(n: u32, low: u32, high: u32) void {
|
||||||
regWrite(redir_base + 2 * n, low);
|
registerWrite(redir_base + 2 * n, low);
|
||||||
regWrite(redir_base + 2 * n + 1, high);
|
registerWrite(redir_base + 2 * n + 1, high);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
||||||
@@ -55,9 +59,9 @@ pub fn init() void {
|
|||||||
// Reach the I/O APIC through the physmap; switch `base` to that virtual
|
// Reach the I/O APIC through the physmap; switch `base` to that virtual
|
||||||
// address so the register accessors work without the identity map.
|
// address so the register accessors work without the identity map.
|
||||||
base = paging.mapMmio(base, 0x1000, true);
|
base = paging.mapMmio(base, 0x1000, true);
|
||||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
maximum_entries = ((registerRead(register_version) >> 16) & 0xFF) + 1;
|
||||||
var n: u32 = 0;
|
var n: u32 = 0;
|
||||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
while (n < maximum_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
||||||
@@ -76,7 +80,7 @@ pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
|||||||
}
|
}
|
||||||
if (gsi < gsi_base) return;
|
if (gsi < gsi_base) return;
|
||||||
const n = gsi - gsi_base;
|
const n = gsi - gsi_base;
|
||||||
if (n >= max_entries) return;
|
if (n >= maximum_entries) return;
|
||||||
|
|
||||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
||||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
||||||
@@ -87,13 +91,62 @@ pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
|||||||
writeEntry(n, low, high);
|
writeEntry(n, low, high);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- GSI-level control (the user-space driver path) --------------------------
|
||||||
|
//
|
||||||
|
// `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides.
|
||||||
|
// A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the
|
||||||
|
// HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection
|
||||||
|
// entry. These three are what `src/kernel/irq.zig` drives.
|
||||||
|
//
|
||||||
|
// Callers must serialise: the I/O APIC is reached through an index/data register
|
||||||
|
// pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel
|
||||||
|
// holds the big lock across these.
|
||||||
|
|
||||||
|
/// Redirection-entry index for `gsi`, or null if this I/O APIC doesn't own it.
|
||||||
|
fn entryFor(gsi: u32) ?u32 {
|
||||||
|
if (base == 0 or gsi < gsi_base) return null;
|
||||||
|
const n = gsi - gsi_base;
|
||||||
|
return if (n < maximum_entries) n else null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if `gsi` lands on this I/O APIC — the kernel's validity check before binding.
|
||||||
|
pub fn ownsGsi(gsi: u32) bool {
|
||||||
|
return entryFor(gsi) != null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Point `gsi` at `vector` on the LAPIC `apic_id`, with explicit polarity/trigger,
|
||||||
|
/// and leave it **masked**. The caller unmasks once a handler is bound — otherwise a
|
||||||
|
/// device asserting between route and bind would fire into a null handler.
|
||||||
|
pub fn routeGsi(gsi: u32, vector: u8, apic_id: u8, level: bool, active_low: bool) void {
|
||||||
|
const n = entryFor(gsi) orelse return;
|
||||||
|
var low: u32 = @as(u32, vector) | redir_mask; // masked until bound
|
||||||
|
if (active_low) low |= (1 << 13);
|
||||||
|
if (level) low |= (1 << 15);
|
||||||
|
writeEntry(n, low, @as(u32, apic_id) << 24);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stop `gsi` reaching any CPU. Called from the ISR *before* the LAPIC EOI: a
|
||||||
|
/// level-triggered line is still asserted at that point, so an unmasked entry would
|
||||||
|
/// redeliver immediately and storm before the user-space driver ever runs.
|
||||||
|
pub fn maskGsi(gsi: u32) void {
|
||||||
|
const n = entryFor(gsi) orelse return;
|
||||||
|
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) | redir_mask);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Let `gsi` through again — the tail of `irq_ack`, once the driver has quieted the
|
||||||
|
/// device (so the line is deasserted and this can't immediately refire).
|
||||||
|
pub fn unmaskGsi(gsi: u32) void {
|
||||||
|
const n = entryFor(gsi) orelse return;
|
||||||
|
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) & ~@as(u32, redir_mask));
|
||||||
|
}
|
||||||
|
|
||||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
||||||
pub fn entryCount() u32 {
|
pub fn entryCount() u32 {
|
||||||
return max_entries;
|
return maximum_entries;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
||||||
pub fn entryLow(n: u32) u32 {
|
pub fn entryLow(n: u32) u32 {
|
||||||
if (base == 0) return 0;
|
if (base == 0) return 0;
|
||||||
return regRead(redir_base + 2 * n);
|
return registerRead(redir_base + 2 * n);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -23,7 +23,7 @@ const pwt: u64 = 1 << 3; // page write-through
|
|||||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||||
const no_execute: u64 = 1 << 63;
|
const no_execute: u64 = 1 << 63;
|
||||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
// ELF segment flags (p_flags).
|
||||||
const pf_x: u32 = 1;
|
const pf_x: u32 = 1;
|
||||||
@@ -54,11 +54,11 @@ const bootstrap_physmap_limit: u64 = 4 << 30;
|
|||||||
/// Dereference a page-table frame by its physical address, via the physmap.
|
/// Dereference a page-table frame by its physical address, via the physmap.
|
||||||
/// This is the single hinge for the higher-half move: page tables hold physical
|
/// This is the single hinge for the higher-half move: page tables hold physical
|
||||||
/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be
|
/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be
|
||||||
/// physical), but the kernel reaches them at `physmap_base + phys`. Valid under
|
/// physical), but the kernel reaches them at `physmap_base + physical`. Valid under
|
||||||
/// both the loader's bootstrap tables and the kernel's own, which share the
|
/// both the loader's bootstrap tables and the kernel's own, which share the
|
||||||
/// physmap base.
|
/// physmap base.
|
||||||
fn tableAt(phys: u64) *[512]u64 {
|
fn tableAt(physical: u64) *[512]u64 {
|
||||||
return @ptrFromInt(danos.physToVirt(phys));
|
return @ptrFromInt(danos.physicalToVirtual(physical));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn allocTable() u64 {
|
fn allocTable() u64 {
|
||||||
@@ -73,41 +73,41 @@ fn allocTable() u64 {
|
|||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||||
/// writable only if every level is; non-executable if any level is).
|
/// writable only if every level is; non-executable if any level is).
|
||||||
fn descend(entry: *u64) u64 {
|
fn descend(entry: *u64) u64 {
|
||||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
if (entry.* & present != 0) return entry.* & address_mask;
|
||||||
const frame = allocTable();
|
const frame = allocTable();
|
||||||
entry.* = frame | present | writable;
|
entry.* = frame | present | writable;
|
||||||
return frame;
|
return frame;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
/// Map one 4 KiB page `virtual` -> `physical` with `flags` (present is added).
|
||||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
// The kernel half is fixed after init: every top-half PML4 entry is
|
// The kernel half is fixed after init: every top-half PML4 entry is
|
||||||
// pre-created so address spaces can share it by copying these slots. A new
|
// pre-created so address spaces can share it by copying these slots. A new
|
||||||
// one here would be invisible to address spaces already made.
|
// one here would be invisible to address spaces already made.
|
||||||
if (init_done and (virt >> 63) == 1 and pml4e.* & present == 0)
|
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||||
@panic("paging: new higher-half PML4 entry after init");
|
@panic("paging: new higher-half PML4 entry after init");
|
||||||
const pdpt = descend(pml4e);
|
const pdpt = descend(pml4e);
|
||||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
const pd = descend(pdpte);
|
const pd = descend(pdpte);
|
||||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
|
||||||
const pt = descend(pde);
|
const pt = descend(pde);
|
||||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map [phys_base, phys_base+len) into the physmap (at physToVirt(phys)) with
|
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||||
/// window onto physical memory once the low identity map goes away.
|
/// window onto physical memory once the low identity map goes away.
|
||||||
fn mapRangePhysmap(pml4: u64, phys_base: u64, len: u64, flags: u64) void {
|
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||||
var addr = phys_base & ~@as(u64, page_size - 1);
|
var address = physical_base & ~@as(u64, page_size - 1);
|
||||||
const end = phys_base + len;
|
const end = physical_base + len;
|
||||||
while (addr < end) : (addr += page_size) {
|
while (address < end) : (address += page_size) {
|
||||||
mapPage(pml4, danos.physToVirt(addr), addr, flags);
|
mapPage(pml4, danos.physicalToVirtual(address), address, flags);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len];
|
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||||
@@ -118,36 +118,36 @@ fn enableNx() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
/// Build the address space and switch onto it.
|
||||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void {
|
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||||
alloc_frame = allocFrame;
|
alloc_frame = allocFrame;
|
||||||
free_frame = freeFrame;
|
free_frame = freeFrame;
|
||||||
enableNx();
|
enableNx();
|
||||||
const pml4 = allocTable();
|
const pml4 = allocTable();
|
||||||
|
|
||||||
// 1. All RAM in the physmap (physToVirt(phys)) RW + NX. No identity/low-half
|
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||||
// mapping: the low half belongs to user space. MMIO is skipped here and
|
// mapping: the low half belongs to user space. MMIO is skipped here and
|
||||||
// mapped on demand (mapMmio) or explicitly below.
|
// mapped on demand (mapMmio) or explicitly below.
|
||||||
for (regions(boot_info.memory_map)) |r| {
|
for (regions(boot_information.memory_map)) |r| {
|
||||||
if (r.kind == .mmio) continue;
|
if (r.kind == .mmio) continue;
|
||||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||||
// the kernel touches directly), RW + NX.
|
// the kernel touches directly), RW + NX.
|
||||||
const fb = boot_info.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||||
mapPage(pml4, danos.physToVirt(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
mapPage(pml4, danos.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||||
|
|
||||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||||
// to their low physical load addresses with real ELF permissions: code
|
// to their low physical load addresses with real ELF permissions: code
|
||||||
// R+X, rodata R, data R+W+NX. This is the W^X guarantee.
|
// R+X, rodata R, data R+W+NX. This is the W^X guarantee.
|
||||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||||
var flags: u64 = present;
|
var flags: u64 = present;
|
||||||
if (seg.flags & pf_w != 0) flags |= writable;
|
if (seg.flags & pf_w != 0) flags |= writable;
|
||||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
if (seg.flags & pf_x == 0) flags |= no_execute;
|
||||||
var off: u64 = 0;
|
var off: u64 = 0;
|
||||||
while (off < seg.pages * page_size) : (off += page_size) {
|
while (off < seg.pages * page_size) : (off += page_size) {
|
||||||
mapPage(pml4, seg.virt + off, seg.phys + off, flags);
|
mapPage(pml4, seg.virtual + off, seg.physical + off, flags);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -187,30 +187,30 @@ pub fn loadCr3(pml4: u64) void {
|
|||||||
|
|
||||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
||||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
/// `writable_page` controls W; pages are always mapped non-executable.
|
||||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
pub fn map(virtual: u64, physical: u64, writable_page: bool) void {
|
||||||
var flags: u64 = present | no_execute;
|
var flags: u64 = present | no_execute;
|
||||||
if (writable_page) flags |= writable;
|
if (writable_page) flags |= writable;
|
||||||
mapPage(kernel_pml4, virt, phys, flags);
|
mapPage(kernel_pml4, virtual, physical, flags);
|
||||||
invalidate(virt);
|
invalidate(virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a device MMIO range into the physmap and return the virtual address to
|
/// Map a device MMIO range into the physmap and return the virtual address to
|
||||||
/// use for it (physToVirt(phys)). The single way the kernel (and the device
|
/// use for it (physicalToVirtual(physical)). The single way the kernel (and the device
|
||||||
/// layer, via the HAL) reaches memory-mapped registers once the identity map is
|
/// layer, via the HAL) reaches memory-mapped registers once the identity map is
|
||||||
/// gone: physmap pages are RW + NX, so a driver never executes device memory.
|
/// gone: physmap pages are RW + NX, so a driver never executes device memory.
|
||||||
/// Idempotent for already-mapped ranges. `len` 0 maps one page.
|
/// Idempotent for already-mapped ranges. `len` 0 maps one page.
|
||||||
pub fn mapMmio(phys: u64, len: u64, writable_page: bool) u64 {
|
pub fn mapMmio(physical: u64, len: u64, writable_page: bool) u64 {
|
||||||
var flags: u64 = present | no_execute;
|
var flags: u64 = present | no_execute;
|
||||||
if (writable_page) flags |= writable;
|
if (writable_page) flags |= writable;
|
||||||
const first = phys & ~@as(u64, page_size - 1);
|
const first = physical & ~@as(u64, page_size - 1);
|
||||||
const last = phys + (if (len == 0) 1 else len) - 1;
|
const last = physical + (if (len == 0) 1 else len) - 1;
|
||||||
var addr = first;
|
var address = first;
|
||||||
while (addr <= (last & ~@as(u64, page_size - 1))) : (addr += page_size) {
|
while (address <= (last & ~@as(u64, page_size - 1))) : (address += page_size) {
|
||||||
const virt = danos.physToVirt(addr);
|
const virtual = danos.physicalToVirtual(address);
|
||||||
mapPage(kernel_pml4, virt, addr, flags);
|
mapPage(kernel_pml4, virtual, address, flags);
|
||||||
invalidate(virt);
|
invalidate(virtual);
|
||||||
}
|
}
|
||||||
return danos.physToVirt(phys);
|
return danos.physicalToVirtual(physical);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
||||||
@@ -223,51 +223,51 @@ fn descendUser(entry: *u64) u64 {
|
|||||||
return table;
|
return table;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map one 4 KiB page `virt` -> `phys` accessible from ring 3. W^X is the
|
/// Map one 4 KiB page `virtual` -> `physical` accessible from ring 3. W^X is the
|
||||||
/// caller's contract: code pages are read-only + executable, data pages are
|
/// caller's contract: code pages are read-only + executable, data pages are
|
||||||
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
|
/// writable + no-execute. `virtual` must lie in a user-exclusive region (see
|
||||||
/// `descendUser`).
|
/// `descendUser`).
|
||||||
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
|
pub fn mapUser(virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
|
||||||
mapUserInto(kernel_pml4, virt, phys, writable_page, executable);
|
mapUserInto(kernel_pml4, virtual, physical, writable_page, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
|
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
|
||||||
/// may be a process's own table or the kernel's). W^X is the caller's contract.
|
/// may be a process's own table or the kernel's). W^X is the caller's contract.
|
||||||
pub fn mapUserInto(pml4: u64, virt: u64, phys: u64, writable_page: bool, executable: bool) void {
|
pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
|
||||||
var flags: u64 = present | user;
|
var flags: u64 = present | user;
|
||||||
if (writable_page) flags |= writable;
|
if (writable_page) flags |= writable;
|
||||||
if (!executable) flags |= no_execute;
|
if (!executable) flags |= no_execute;
|
||||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
const pdpt = descendUser(pml4e);
|
const pdpt = descendUser(pml4e);
|
||||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
const pd = descendUser(pdpte);
|
const pd = descendUser(pdpte);
|
||||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
|
||||||
const pt = descendUser(pde);
|
const pt = descendUser(pde);
|
||||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags;
|
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags;
|
||||||
invalidate(virt);
|
invalidate(virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a device MMIO window `[phys, phys+len)` into the user (low) half of the
|
/// Map a device MMIO window `[physical, physical+len)` into the user (low) half of the
|
||||||
/// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages
|
/// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages
|
||||||
/// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and
|
/// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and
|
||||||
/// carry the `device_grant` bit so teardown does not return the MMIO frames to the
|
/// carry the `device_grant` bit so teardown does not return the MMIO frames to the
|
||||||
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virt` in a
|
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
||||||
/// user-exclusive range (PML4[225]). Both `virt` and `phys` are page-aligned by
|
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
||||||
/// the caller; a sub-page `phys` offset is the caller's to re-apply.
|
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
||||||
pub fn mapUserDeviceInto(pml4: u64, virt: u64, phys: u64, len: u64) void {
|
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||||
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
||||||
const first = phys & ~@as(u64, page_size - 1);
|
const first = physical & ~@as(u64, page_size - 1);
|
||||||
const last = (phys + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||||
var off: u64 = 0;
|
var off: u64 = 0;
|
||||||
while (first + off <= last) : (off += page_size) {
|
while (first + off <= last) : (off += page_size) {
|
||||||
const v = virt + off;
|
const v = virtual + off;
|
||||||
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||||
const pdpt = descendUser(pml4e);
|
const pdpt = descendUser(pml4e);
|
||||||
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||||
const pd = descendUser(pdpte);
|
const pd = descendUser(pdpte);
|
||||||
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||||
const pt = descendUser(pde);
|
const pt = descendUser(pde);
|
||||||
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & addr_mask) | flags;
|
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||||
invalidate(v);
|
invalidate(v);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -291,39 +291,39 @@ pub fn createAddressSpace() ?u64 {
|
|||||||
pub fn destroyAddressSpace(pml4: u64) void {
|
pub fn destroyAddressSpace(pml4: u64) void {
|
||||||
const t = tableAt(pml4);
|
const t = tableAt(pml4);
|
||||||
for (0..256) |i| {
|
for (0..256) |i| {
|
||||||
if (t[i] & present != 0) freeSubtree(t[i] & addr_mask, 3); // PDPT level
|
if (t[i] & present != 0) freeSubtree(t[i] & address_mask, 3); // PDPT level
|
||||||
}
|
}
|
||||||
free_frame(pml4);
|
free_frame(pml4);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
|
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
|
||||||
/// level 1 the entries are leaf data frames; above, they are child tables.
|
/// level 1 the entries are leaf data frames; above, they are child tables.
|
||||||
fn freeSubtree(phys: u64, level: u32) void {
|
fn freeSubtree(physical: u64, level: u32) void {
|
||||||
const t = tableAt(phys);
|
const t = tableAt(physical);
|
||||||
for (t) |e| {
|
for (t) |e| {
|
||||||
if (e & present == 0) continue;
|
if (e & present == 0) continue;
|
||||||
if (level > 1) {
|
if (level > 1) {
|
||||||
freeSubtree(e & addr_mask, level - 1);
|
freeSubtree(e & address_mask, level - 1);
|
||||||
} else if (e & device_grant == 0) {
|
} else if (e & device_grant == 0) {
|
||||||
// A device-grant leaf points at MMIO, not RAM — returning it to the
|
// A device-grant leaf points at MMIO, not RAM — returning it to the
|
||||||
// frame allocator would corrupt the pool. Only reclaim real RAM.
|
// frame allocator would corrupt the pool. Only reclaim real RAM.
|
||||||
free_frame(e & addr_mask);
|
free_frame(e & address_mask);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
free_frame(phys); // page-table frames are always real RAM
|
free_frame(physical); // page-table frames are always real RAM
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether `virt` is currently mapped **executable** — present with the NX bit
|
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||||
pub fn isExecutable(virt: u64) bool {
|
pub fn isExecutable(virtual: u64) bool {
|
||||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return false;
|
if (pml4e & present == 0) return false;
|
||||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
if (pdpte & present == 0) return false;
|
if (pdpte & present == 0) return false;
|
||||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return false;
|
if (pde & present == 0) return false;
|
||||||
const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return false;
|
if (pte & present == 0) return false;
|
||||||
return pte & no_execute == 0;
|
return pte & no_execute == 0;
|
||||||
}
|
}
|
||||||
@@ -333,57 +333,57 @@ pub fn isExecutable(virt: u64) bool {
|
|||||||
/// application processors fetch the AP trampoline from a low RAM page under paging —
|
/// application processors fetch the AP trampoline from a low RAM page under paging —
|
||||||
/// so that one page must be executable. A deliberate, temporary W^X exception for a
|
/// so that one page must be executable. A deliberate, temporary W^X exception for a
|
||||||
/// single bring-up page; the caller frees it once every AP is up.
|
/// single bring-up page; the caller frees it once every AP is up.
|
||||||
pub fn setExecutable(phys: u64) void {
|
pub fn setExecutable(physical: u64) void {
|
||||||
mapPage(kernel_pml4, phys, phys, present | writable); // note: no no_execute
|
mapPage(kernel_pml4, physical, physical, present | writable); // note: no no_execute
|
||||||
invalidate(phys);
|
invalidate(physical);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Remove a mapping and flush it from the TLB.
|
/// Remove a mapping and flush it from the TLB.
|
||||||
pub fn unmap(virt: u64) void {
|
pub fn unmap(virtual: u64) void {
|
||||||
unmapInto(kernel_pml4, virt);
|
unmapInto(kernel_pml4, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Remove a mapping from the address space rooted at `pml4` (a process's own
|
/// Remove a mapping from the address space rooted at `pml4` (a process's own
|
||||||
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
|
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
|
||||||
/// the intermediate tables and any frame the PTE pointed at are left to the
|
/// the intermediate tables and any frame the PTE pointed at are left to the
|
||||||
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
|
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
|
||||||
pub fn unmapInto(pml4: u64, virt: u64) void {
|
pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||||
const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return;
|
if (pml4e & present == 0) return;
|
||||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
if (pdpte & present == 0) return;
|
if (pdpte & present == 0) return;
|
||||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return;
|
if (pde & present == 0) return;
|
||||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF] = 0;
|
||||||
invalidate(virt);
|
invalidate(virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||||
/// any address space, not just the live one). Returns null if `virt` is not
|
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||||
/// needs the frame behind a user vaddr to free it).
|
/// needs the frame behind a user vaddr to free it).
|
||||||
pub fn translateIn(pml4: u64, virt: u64) ?u64 {
|
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||||
const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return null;
|
if (pml4e & present == 0) return null;
|
||||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
if (pdpte & present == 0) return null;
|
if (pdpte & present == 0) return null;
|
||||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return null;
|
if (pde & present == 0) return null;
|
||||||
const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return null;
|
if (pte & present == 0) return null;
|
||||||
return (pte & addr_mask) | (virt & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn invalidate(virt: u64) void {
|
fn invalidate(virtual: u64) void {
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
// inline asm won't form directly, so stage the address in a register first.
|
||||||
asm volatile (
|
asm volatile (
|
||||||
\\mov %[v], %%rax
|
\\mov %[v], %%rax
|
||||||
\\invlpg (%%rax)
|
\\invlpg (%%rax)
|
||||||
:
|
:
|
||||||
: [v] "r" (virt),
|
: [v] "r" (virtual),
|
||||||
: .{ .rax = true, .memory = true }
|
: .{ .rax = true, .memory = true }
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,74 +1,74 @@
|
|||||||
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
|
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
|
||||||
//! to this core's `ArchPerCpu`, so kernel code gets the running core's block with
|
//! to this core's `ArchitecturePerCpu`, so kernel code gets the running core's block with
|
||||||
//! a single MSR read (`sched()`) and the syscall entry stub gets its kernel stack
|
//! a single MSR read (`scheduler()`) and the system_call entry stub gets its kernel stack
|
||||||
//! with a `%gs`-relative load (no usable stack yet at that point).
|
//! with a `%gs`-relative load (no usable stack yet at that point).
|
||||||
//!
|
//!
|
||||||
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
|
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
|
||||||
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
|
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
|
||||||
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
|
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
|
||||||
//! syscall stub and the conditional swapgs in isr_common) brings it back, and
|
//! system_call stub and the conditional swapgs in isr_common) brings it back, and
|
||||||
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
|
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
|
||||||
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
|
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
|
||||||
//! keep the invariant without seeding KERNEL_GS_BASE. `sched()` is therefore
|
//! keep the invariant without seeding KERNEL_GS_BASE. `scheduler()` is therefore
|
||||||
//! valid in any ring-0 context and never sees a user-controlled base.
|
//! valid in any ring-0 context and never sees a user-controlled base.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const io = @import("io.zig");
|
const io = @import("io.zig");
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
|
|
||||||
const ia32_gs_base = 0xC000_0101;
|
const ia32_gs_base = 0xC000_0101;
|
||||||
|
|
||||||
/// Layout is load-bearing: the syscall entry stub in isr.s reaches `kernel_rsp`
|
/// Layout is load-bearing: the system_call entry stub in isr.s reaches `kernel_rsp`
|
||||||
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
|
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
|
||||||
/// pin the offsets.
|
/// pin the offsets.
|
||||||
pub const ArchPerCpu = extern struct {
|
pub const ArchitecturePerCpu = extern struct {
|
||||||
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for syscall entry (== TSS.rsp0)
|
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for system_call entry (== TSS.rsp0)
|
||||||
scratch: u64 = 0, // %gs:8 — stashes the user rsp during syscall entry
|
scratch: u64 = 0, // %gs:8 — stashes the user rsp during system_call entry
|
||||||
sched: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
|
scheduler: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
|
||||||
};
|
};
|
||||||
|
|
||||||
comptime {
|
comptime {
|
||||||
std.debug.assert(@offsetOf(ArchPerCpu, "kernel_rsp") == 0);
|
std.debug.assert(@offsetOf(ArchitecturePerCpu, "kernel_rsp") == 0);
|
||||||
std.debug.assert(@offsetOf(ArchPerCpu, "scratch") == 8);
|
std.debug.assert(@offsetOf(ArchitecturePerCpu, "scratch") == 8);
|
||||||
}
|
}
|
||||||
|
|
||||||
var blocks = [_]ArchPerCpu{.{}} ** config.max_cpus;
|
var blocks = [_]ArchitecturePerCpu{.{}} ** parameters.maximum_cpus;
|
||||||
|
|
||||||
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
|
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
|
||||||
/// the GS base at the block. Called once per core during bring-up, after the GDT
|
/// the GS base at the block. Called once per core during bring-up, after the GDT
|
||||||
/// is loaded (a GS *selector* reload would clobber the base).
|
/// is loaded (a GS *selector* reload would clobber the base).
|
||||||
pub fn setLocal(index: usize, sched_ptr: usize) void {
|
pub fn setLocal(index: usize, scheduler_ptr: usize) void {
|
||||||
blocks[index].sched = sched_ptr;
|
blocks[index].scheduler = scheduler_ptr;
|
||||||
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
|
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The scheduler pointer for the running core (via the GS base). Valid in any
|
/// The scheduler pointer for the running core (via the GS base). Valid in any
|
||||||
/// ring-0 context under the swapgs discipline.
|
/// ring-0 context under the swapgs discipline.
|
||||||
pub fn sched() usize {
|
pub fn scheduler() usize {
|
||||||
return @as(*const ArchPerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).sched;
|
return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Record core `index`'s kernel stack top, used by the syscall entry stub to
|
/// Record core `index`'s kernel stack top, used by the system_call entry stub to
|
||||||
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
|
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
|
||||||
/// switches to a user task.
|
/// switches to a user task.
|
||||||
pub fn setKernelRsp(index: usize, top: usize) void {
|
pub fn setKernelRsp(index: usize, top: usize) void {
|
||||||
blocks[index].kernel_rsp = top;
|
blocks[index].kernel_rsp = top;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Fast-syscall MSRs.
|
// Fast-system_call MSRs.
|
||||||
const ia32_efer = 0xC000_0080;
|
const ia32_efer = 0xC000_0080;
|
||||||
const ia32_star = 0xC000_0081;
|
const ia32_star = 0xC000_0081;
|
||||||
const ia32_lstar = 0xC000_0082;
|
const ia32_lstar = 0xC000_0082;
|
||||||
const ia32_sfmask = 0xC000_0084;
|
const ia32_sfmask = 0xC000_0084;
|
||||||
|
|
||||||
/// Enable the `syscall`/`sysret` fast path on this core (BSP and each AP). EFER.SCE
|
/// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE
|
||||||
/// turns the instructions on; STAR sets the selectors syscall/sysret load; LSTAR
|
/// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR
|
||||||
/// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF —
|
/// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF —
|
||||||
/// the handler runs with interrupts off, like the int-gate path). The GDT is laid
|
/// the handler runs with interrupts off, like the int-gate path). The GDT is laid
|
||||||
/// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line
|
/// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line
|
||||||
/// up: syscall loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8
|
/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8
|
||||||
/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B.
|
/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B.
|
||||||
pub fn initSyscall() void {
|
pub fn initSystemCall() void {
|
||||||
io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE
|
io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE
|
||||||
io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48));
|
io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48));
|
||||||
const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" });
|
const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" });
|
||||||
@@ -33,13 +33,13 @@ fn portIn(p: u16) u8 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read UART register `off` through the active access method.
|
/// Read UART register `off` through the active access method.
|
||||||
fn reg(off: u64) u8 {
|
fn register(off: u64) u8 {
|
||||||
if (access == .mmio) return @as(*volatile u8, @ptrFromInt(base + off)).*;
|
if (access == .mmio) return @as(*volatile u8, @ptrFromInt(base + off)).*;
|
||||||
return portIn(@intCast(base + off));
|
return portIn(@intCast(base + off));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write UART register `off` through the active access method.
|
/// Write UART register `off` through the active access method.
|
||||||
fn setReg(off: u64, value: u8) void {
|
fn setRegister(off: u64, value: u8) void {
|
||||||
if (access == .mmio) {
|
if (access == .mmio) {
|
||||||
@as(*volatile u8, @ptrFromInt(base + off)).* = value;
|
@as(*volatile u8, @ptrFromInt(base + off)).* = value;
|
||||||
} else {
|
} else {
|
||||||
@@ -50,22 +50,22 @@ fn setReg(off: u64, value: u8) void {
|
|||||||
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
|
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
|
||||||
/// else; it has no dependencies, and is a harmless no-op if the port is absent.
|
/// else; it has no dependencies, and is a harmless no-op if the port is absent.
|
||||||
pub fn init() void {
|
pub fn init() void {
|
||||||
setReg(1, 0x00); // disable interrupts
|
setRegister(1, 0x00); // disable interrupts
|
||||||
setReg(3, 0x80); // enable DLAB (set baud divisor)
|
setRegister(3, 0x80); // enable DLAB (set baud divisor)
|
||||||
setReg(0, 0x03); // divisor low: 38400 baud
|
setRegister(0, 0x03); // divisor low: 38400 baud
|
||||||
setReg(1, 0x00); // divisor high
|
setRegister(1, 0x00); // divisor high
|
||||||
setReg(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||||
setReg(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||||
setReg(4, 0x0B); // RTS/DSR set
|
setRegister(4, 0x0B); // RTS/DSR set
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||||
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
|
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
|
||||||
pub fn reconfigure(is_mmio: bool, addr: u64) void {
|
pub fn reconfigure(is_mmio: bool, address: u64) void {
|
||||||
access = if (is_mmio) .mmio else .port;
|
access = if (is_mmio) .mmio else .port;
|
||||||
// An MMIO UART is reached through the physmap; an I/O-port UART keeps its
|
// An MMIO UART is reached through the physmap; an I/O-port UART keeps its
|
||||||
// port number unchanged.
|
// port number unchanged.
|
||||||
base = if (is_mmio) paging.mapMmio(addr, 0x100, true) else addr;
|
base = if (is_mmio) paging.mapMmio(address, 0x100, true) else address;
|
||||||
init();
|
init();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -73,8 +73,8 @@ fn writeByte(c: u8) void {
|
|||||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
||||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
||||||
var guard: u32 = 0;
|
var guard: u32 = 0;
|
||||||
while (reg(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
||||||
setReg(0, c);
|
setRegister(0, c);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ const tss = @import("tss.zig");
|
|||||||
const idt = @import("idt.zig");
|
const idt = @import("idt.zig");
|
||||||
const apic = @import("apic.zig");
|
const apic = @import("apic.zig");
|
||||||
const paging = @import("paging.zig");
|
const paging = @import("paging.zig");
|
||||||
const pcpu = @import("percpu.zig");
|
const pcpu = @import("per-cpu.zig");
|
||||||
|
|
||||||
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
|
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
|
||||||
/// path doesn't depend on cpu.zig and risk an import cycle).
|
/// path doesn't depend on cpu.zig and risk an import cycle).
|
||||||
@@ -31,9 +31,9 @@ const page_size = 0x1000;
|
|||||||
/// the life of the system so any core can be (re)woken on demand — a retry, or a
|
/// the life of the system so any core can be (re)woken on demand — a retry, or a
|
||||||
/// future power manager bringing a core back online. The frame is kept **inert**
|
/// future power manager bringing a core back online. The frame is kept **inert**
|
||||||
/// between wakes (zeroed and non-executable) and only armed for the brief moment a
|
/// between wakes (zeroed and non-executable) and only armed for the brief moment a
|
||||||
/// core is actually climbing. Its low 20 bits are zero, so `phys >> 12` is the SIPI
|
/// core is actually climbing. Its low 20 bits are zero, so `physical >> 12` is the SIPI
|
||||||
/// vector.
|
/// vector.
|
||||||
var tramp_phys: u64 = 0;
|
var tramp_physical: u64 = 0;
|
||||||
|
|
||||||
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
|
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
|
||||||
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
|
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
|
||||||
@@ -44,7 +44,7 @@ var ap_alive: u32 = 0;
|
|||||||
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
|
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
|
||||||
var boot_index: usize = 0;
|
var boot_index: usize = 0;
|
||||||
|
|
||||||
/// The generic scheduler entry a woken core jumps to once its arch state is up. Set
|
/// The generic scheduler entry a woken core jumps to once its architecture state is up. Set
|
||||||
/// by the kernel via `setSecondaryEntry`; never returns.
|
/// by the kernel via `setSecondaryEntry`; never returns.
|
||||||
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
|
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
|
||||||
|
|
||||||
@@ -55,7 +55,7 @@ pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
|||||||
|
|
||||||
/// Test hook: force the next `n` wake attempts to fail (skipping the actual
|
/// Test hook: force the next `n` wake attempts to fail (skipping the actual
|
||||||
/// INIT-SIPI-SIPI), so the retry path can be exercised deterministically. Zero in
|
/// INIT-SIPI-SIPI), so the retry path can be exercised deterministically. Zero in
|
||||||
/// normal operation — the smp-retry test arms it via `arch.testFailNextWakes`.
|
/// normal operation — the smp-retry test arms it via `architecture.testFailNextWakes`.
|
||||||
var fail_next_wakes: u32 = 0;
|
var fail_next_wakes: u32 = 0;
|
||||||
pub fn testFailNextWakes(n: u32) void {
|
pub fn testFailNextWakes(n: u32) void {
|
||||||
fail_next_wakes = n;
|
fail_next_wakes = n;
|
||||||
@@ -64,14 +64,14 @@ pub fn testFailNextWakes(n: u32) void {
|
|||||||
/// Record the reserved low frame the trampoline uses. Call once at boot. The frame
|
/// Record the reserved low frame the trampoline uses. Call once at boot. The frame
|
||||||
/// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms
|
/// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms
|
||||||
/// it again, so it's only ever executable while a core is climbing.
|
/// it again, so it's only ever executable while a core is climbing.
|
||||||
pub fn setTrampolinePage(phys: u64) void {
|
pub fn setTrampolinePage(physical: u64) void {
|
||||||
tramp_phys = phys;
|
tramp_physical = physical;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The reserved trampoline frame (0 if SMP bring-up never ran). Exposed so a test
|
/// The reserved trampoline frame (0 if SMP bring-up never ran). Exposed so a test
|
||||||
/// can verify it's inert — zeroed and non-executable — when dormant.
|
/// can verify it's inert — zeroed and non-executable — when dormant.
|
||||||
pub fn trampolinePage() u64 {
|
pub fn trampolinePage() u64 {
|
||||||
return tramp_phys;
|
return tramp_physical;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
|
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
|
||||||
@@ -81,12 +81,12 @@ fn arm() void {
|
|||||||
// climbs from real to long mode, so it needs a low identity mapping that is
|
// climbs from real to long mode, so it needs a low identity mapping that is
|
||||||
// executable — the one deliberate, transient W^X exception. The BSP writes
|
// executable — the one deliberate, transient W^X exception. The BSP writes
|
||||||
// the blob into the frame through the physmap.
|
// the blob into the frame through the physmap.
|
||||||
paging.setExecutable(tramp_phys);
|
paging.setExecutable(tramp_physical);
|
||||||
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
||||||
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
||||||
const len = @intFromPtr(end) - @intFromPtr(start);
|
const len = @intFromPtr(end) - @intFromPtr(start);
|
||||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys));
|
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||||
@memcpy(dst[0..len], start[0..len]);
|
@memcpy(destination[0..len], start[0..len]);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Disarm after a wake: wipe the page through the physmap and remove its low
|
/// Disarm after a wake: wipe the page through the physmap and remove its low
|
||||||
@@ -95,9 +95,9 @@ fn arm() void {
|
|||||||
/// reported in — it's long past the trampoline by then, in the kernel image; a
|
/// reported in — it's long past the trampoline by then, in the kernel image; a
|
||||||
/// core that never answered is dead and can't be mid-climb.
|
/// core that never answered is dead and can't be mid-climb.
|
||||||
fn disarm() void {
|
fn disarm() void {
|
||||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys));
|
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||||
@memset(dst[0..page_size], 0);
|
@memset(destination[0..page_size], 0);
|
||||||
paging.unmap(tramp_phys); // drop the transient low identity mapping
|
paging.unmap(tramp_physical); // drop the transient low identity mapping
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
|
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
|
||||||
@@ -107,7 +107,7 @@ fn disarm() void {
|
|||||||
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
||||||
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
||||||
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
||||||
return @ptrFromInt(danos.physToVirt(tramp_phys + (sym - start)));
|
return @ptrFromInt(danos.physicalToVirtual(tramp_physical + (sym - start)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
||||||
@@ -138,7 +138,7 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3:
|
|||||||
|
|
||||||
@atomicStore(u32, &ap_alive, 0, .seq_cst);
|
@atomicStore(u32, &ap_alive, 0, .seq_cst);
|
||||||
|
|
||||||
const vector: u8 = @intCast(tramp_phys >> 12);
|
const vector: u8 = @intCast(tramp_physical >> 12);
|
||||||
apic.sendInit(apic_id);
|
apic.sendInit(apic_id);
|
||||||
delayMicros(10_000); // 10 ms INIT settle
|
delayMicros(10_000); // 10 ms INIT settle
|
||||||
apic.sendStartup(apic_id, vector);
|
apic.sendStartup(apic_id, vector);
|
||||||
@@ -170,12 +170,12 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
|
|||||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||||
idt.loadOnThisCpu(); // the shared IDT
|
idt.loadOnThisCpu(); // the shared IDT
|
||||||
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
|
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
|
||||||
pcpu.initSyscall(); // enable syscall/sysret on this core
|
pcpu.initSystemCall(); // enable system_call/sysret on this core
|
||||||
|
|
||||||
apic.initSecondary(); // software-enable this core's LAPIC
|
apic.initSecondary(); // software-enable this core's LAPIC
|
||||||
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
|
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
|
||||||
|
|
||||||
@atomicStore(u32, &ap_alive, 1, .release); // "arch state up" — BSP is polling this
|
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||||
|
|
||||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||||
|
|||||||
@@ -11,7 +11,7 @@
|
|||||||
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
|
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
|
||||||
//! indexed by CPU number; slot 0 is the BSP.
|
//! indexed by CPU number; slot 0 is the BSP.
|
||||||
|
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
const gdt = @import("gdt.zig");
|
const gdt = @import("gdt.zig");
|
||||||
|
|
||||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
||||||
@@ -37,17 +37,17 @@ const Tss = packed struct {
|
|||||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
||||||
pub const double_fault_ist = 1;
|
pub const double_fault_ist = 1;
|
||||||
|
|
||||||
const max_cpus = config.max_cpus;
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
pub const ist_stack_size = config.ist_stack_size;
|
pub const ist_stack_size = parameters.ist_stack_size;
|
||||||
|
|
||||||
/// One TSS per core (small — kept static). The IST stacks are 16 KiB each, so only
|
/// One TSS per core (small — kept static). The IST stacks are 16 KiB each, so only
|
||||||
/// the **BSP's** is static: it must exist before the frame allocator does, to catch a
|
/// the **BSP's** is static: it must exist before the frame allocator does, to catch a
|
||||||
/// fault during early boot. Each **AP** gets a heap-allocated IST stack at bring-up
|
/// fault during early boot. Each **AP** gets a heap-allocated IST stack at bring-up
|
||||||
/// (after the heap is up), the top of which the BSP records here before waking it —
|
/// (after the heap is up), the top of which the BSP records here before waking it —
|
||||||
/// so we reserve big stacks only for cores that actually come online.
|
/// so we reserve big stacks only for cores that actually come online.
|
||||||
var tss_table = [_]Tss{.{}} ** max_cpus;
|
var tss_table = [_]Tss{.{}} ** maximum_cpus;
|
||||||
var bsp_ist_stack: [ist_stack_size]u8 align(16) = undefined;
|
var bsp_ist_stack: [ist_stack_size]u8 align(16) = undefined;
|
||||||
var ap_ist_top = [_]usize{0} ** max_cpus; // per-AP IST stack top (0 = BSP / not set)
|
var ap_ist_top = [_]usize{0} ** maximum_cpus; // per-AP IST stack top (0 = BSP / not set)
|
||||||
|
|
||||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||||
|
|||||||
@@ -71,7 +71,7 @@ pub const Console = struct {
|
|||||||
// once the low identity map is gone. The base is mapped by both the
|
// once the low identity map is gone. The base is mapped by both the
|
||||||
// loader's bootstrap tables and paging.init.
|
// loader's bootstrap tables and paging.init.
|
||||||
var mapped = fb;
|
var mapped = fb;
|
||||||
if (fb.base != 0) mapped.base = danos.physToVirt(fb.base);
|
if (fb.base != 0) mapped.base = danos.physicalToVirtual(fb.base);
|
||||||
return .{
|
return .{
|
||||||
.fb = mapped,
|
.fb = mapped,
|
||||||
.cols = fb.width / glyph_w,
|
.cols = fb.width / glyph_w,
|
||||||
@@ -147,11 +147,11 @@ pub const Console = struct {
|
|||||||
while (x < self.fb.width) : (x += 1) row[x] = color;
|
while (x < self.fb.width) : (x += 1) row[x] = color;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn copyRow(self: *Console, dst_y: u32, src_y: u32) void {
|
fn copyRow(self: *Console, destination_y: u32, source_y: u32) void {
|
||||||
const dst = self.rowPtr(dst_y);
|
const destination = self.rowPtr(destination_y);
|
||||||
const src = self.rowPtr(src_y);
|
const source = self.rowPtr(source_y);
|
||||||
var x: u32 = 0;
|
var x: u32 = 0;
|
||||||
while (x < self.fb.width) : (x += 1) dst[x] = src[x];
|
while (x < self.fb.width) : (x += 1) destination[x] = source[x];
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,181 @@
|
|||||||
|
//! Device service: the kernel side of user-space driver access. At boot it
|
||||||
|
//! flattens the discovered device tree (src/device) into a stable, id-indexed
|
||||||
|
//! snapshot and a per-device claim table. User drivers enumerate the snapshot,
|
||||||
|
//! claim the device they own, and map its MMIO — the claim is the capability that
|
||||||
|
//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the
|
||||||
|
//! firmware-neutral device tree says it owns.
|
||||||
|
//!
|
||||||
|
//! The table is a **tree**: each entry carries its parent's id. Firmware discovery
|
||||||
|
//! seeds it, and a **bus driver** grows it — a process that has claimed a bus can
|
||||||
|
//! `register` children below it as it enumerates them (USB devices behind a hub, PCI
|
||||||
|
//! functions behind a bridge, comparators inside a timer block).
|
||||||
|
//!
|
||||||
|
//! Registration is where the capability model earns its keep. A `DeviceDescriptor` is, in
|
||||||
|
//! effect, a licence to map physical memory: whoever claims it may `mmio_map` its
|
||||||
|
//! `.memory` resources and `irq_bind` its `.irq` resources. If a bus driver could
|
||||||
|
//! invent arbitrary resources, it would invent one covering the kernel's RAM, claim
|
||||||
|
//! it, and map it. So `register` enforces **containment**: every resource of a child
|
||||||
|
//! must lie inside a resource of the same kind on its parent. A bus driver can only
|
||||||
|
//! ever subdivide what it was already given.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const platform = @import("platform");
|
||||||
|
const danos = @import("danos");
|
||||||
|
|
||||||
|
const maximum_devices = 64;
|
||||||
|
|
||||||
|
/// Cap on children a single parent may have. A zero-resource child (legal — a USB
|
||||||
|
/// device is addressed through its controller, not by MMIO) sidesteps the containment
|
||||||
|
/// check, so without a bound a process that claimed one device could loop
|
||||||
|
/// `device_register` and exhaust the whole table, permanently denying it to every other
|
||||||
|
/// driver. This bounds the blast radius of one claim; a real quota (and a
|
||||||
|
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
|
||||||
|
const maximum_children_per_parent = 16;
|
||||||
|
|
||||||
|
var devices: [maximum_devices]danos.DeviceDescriptor = undefined;
|
||||||
|
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||||
|
var count: usize = 0;
|
||||||
|
|
||||||
|
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||||
|
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||||
|
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||||
|
pub var dropped: usize = 0;
|
||||||
|
|
||||||
|
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
||||||
|
pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||||
|
count = 0;
|
||||||
|
dropped = 0;
|
||||||
|
for (&claimed) |*c| c.* = null;
|
||||||
|
walk(device_tree.root, danos.no_parent);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||||
|
/// assigned it down to its children as their parent.
|
||||||
|
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||||
|
const id = if (node.class == .root) danos.no_parent else record(node, parent_id);
|
||||||
|
var child = node.first_child;
|
||||||
|
while (child) |c| : (child = c.next_sibling) walk(c, id);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||||
|
if (count >= maximum_devices) {
|
||||||
|
dropped += 1;
|
||||||
|
return danos.no_parent; // children of a dropped node become roots, not orphans
|
||||||
|
}
|
||||||
|
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||||
|
d.id = count;
|
||||||
|
d.parent = parent_id;
|
||||||
|
d.class = @intFromEnum(node.class);
|
||||||
|
const h = node.hid();
|
||||||
|
d.hid_len = @min(h.len, d.hid.len);
|
||||||
|
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
|
||||||
|
const rc = @min(node.resource_count, danos.maximum_device_resources);
|
||||||
|
d.resource_count = rc;
|
||||||
|
for (0..rc) |i| {
|
||||||
|
const r = node.resources[i];
|
||||||
|
d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len };
|
||||||
|
}
|
||||||
|
devices[count] = d;
|
||||||
|
count += 1;
|
||||||
|
return d.id;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||||
|
/// available (which may exceed `out.len`).
|
||||||
|
pub fn enumerate(out: []danos.DeviceDescriptor) usize {
|
||||||
|
const n = @min(count, out.len);
|
||||||
|
@memcpy(out[0..n], devices[0..n]);
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||||
|
/// out of range or already claimed.
|
||||||
|
pub fn claim(id: u64, owner: u32) bool {
|
||||||
|
if (id >= count) return false;
|
||||||
|
if (claimed[@intCast(id)] != null) return false;
|
||||||
|
claimed[@intCast(id)] = owner;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The task that owns device `id`, or null.
|
||||||
|
pub fn ownerOf(id: u64) ?u32 {
|
||||||
|
if (id >= count) return null;
|
||||||
|
return claimed[@intCast(id)];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resource `index` of device `id`, or null if out of range.
|
||||||
|
pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor {
|
||||||
|
if (id >= count) return null;
|
||||||
|
const d = &devices[@intCast(id)];
|
||||||
|
if (index >= d.resource_count) return null;
|
||||||
|
return d.resources[@intCast(index)];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||||
|
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||||
|
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||||
|
/// and would otherwise vacuously "fit" anywhere.
|
||||||
|
fn contains(parent: danos.ResourceDescriptor, child: danos.ResourceDescriptor) bool {
|
||||||
|
if (parent.kind != child.kind) return false;
|
||||||
|
if (child.kind == @intFromEnum(danos.ResourceKind.irq)) return parent.start == child.start;
|
||||||
|
if (child.len == 0 or parent.len == 0) return false;
|
||||||
|
// No overflow: a resource that wraps the address space is not containable.
|
||||||
|
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
||||||
|
const parent_end = std.math.add(u64, parent.start, parent.len) catch return false;
|
||||||
|
return child.start >= parent.start and child_end <= parent_end;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const RegisterError = error{
|
||||||
|
NoSpace, // the device table is full
|
||||||
|
BadParent, // no such device, or not claimed by this task
|
||||||
|
TooManyResources,
|
||||||
|
TooManyChildren, // this parent is at maximum_children_per_parent
|
||||||
|
NotContained, // a child resource escapes its parent's window
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Number of devices currently recorded with `parent_id` as their parent.
|
||||||
|
fn childCount(parent_id: u64) usize {
|
||||||
|
var n: usize = 0;
|
||||||
|
for (devices[0..count]) |d| {
|
||||||
|
if (d.parent == parent_id) n += 1;
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
||||||
|
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
||||||
|
/// can claim it — that is how a bus hands a device to its driver.
|
||||||
|
///
|
||||||
|
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
|
||||||
|
/// contained in a parent resource of the same kind. A device with no resources is
|
||||||
|
/// fine and common: a USB device is addressed through its controller, not by MMIO.
|
||||||
|
pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescriptor) RegisterError!u64 {
|
||||||
|
const parent_owner = ownerOf(parent_id) orelse return error.BadParent;
|
||||||
|
if (parent_owner != owner) return error.BadParent;
|
||||||
|
if (descriptor.resource_count > danos.maximum_device_resources) return error.TooManyResources;
|
||||||
|
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
|
||||||
|
if (count >= maximum_devices) return error.NoSpace;
|
||||||
|
|
||||||
|
const parent = &devices[@intCast(parent_id)];
|
||||||
|
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||||
|
const r = descriptor.resources[i];
|
||||||
|
var ok = false;
|
||||||
|
for (0..@intCast(parent.resource_count)) |j| {
|
||||||
|
if (contains(parent.resources[j], r)) ok = true;
|
||||||
|
}
|
||||||
|
if (!ok) return error.NotContained;
|
||||||
|
}
|
||||||
|
|
||||||
|
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||||
|
d.id = count;
|
||||||
|
d.parent = parent_id;
|
||||||
|
d.class = descriptor.class;
|
||||||
|
d.hid_len = @min(descriptor.hid_len, d.hid.len);
|
||||||
|
@memcpy(d.hid[0..@intCast(d.hid_len)], descriptor.hid[0..@intCast(d.hid_len)]);
|
||||||
|
d.resource_count = descriptor.resource_count;
|
||||||
|
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
|
||||||
|
|
||||||
|
devices[count] = d;
|
||||||
|
count += 1;
|
||||||
|
return d.id;
|
||||||
|
}
|
||||||
@@ -1,78 +0,0 @@
|
|||||||
//! Device service: the kernel side of user-space driver access. At boot it
|
|
||||||
//! flattens the discovered device tree (src/device) into a stable, id-indexed
|
|
||||||
//! snapshot and a per-device claim table. User drivers enumerate the snapshot,
|
|
||||||
//! claim the device they own, and map its MMIO — the claim is the capability that
|
|
||||||
//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the
|
|
||||||
//! firmware-neutral device tree says it owns.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const danos = @import("danos");
|
|
||||||
|
|
||||||
const max_devices = 32;
|
|
||||||
|
|
||||||
var devices: [max_devices]danos.DeviceDesc = undefined;
|
|
||||||
var claimed: [max_devices]?u32 = .{null} ** max_devices; // owner task id, or null
|
|
||||||
var count: usize = 0;
|
|
||||||
|
|
||||||
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
|
||||||
pub fn init(dt: *const platform.DeviceTree) void {
|
|
||||||
count = 0;
|
|
||||||
for (&claimed) |*c| c.* = null;
|
|
||||||
walk(dt.root);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn walk(node: *platform.Device) void {
|
|
||||||
if (node.class != .root) record(node);
|
|
||||||
var child = node.first_child;
|
|
||||||
while (child) |c| : (child = c.next_sibling) walk(c);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn record(node: *platform.Device) void {
|
|
||||||
if (count >= max_devices) return;
|
|
||||||
var d = std.mem.zeroes(danos.DeviceDesc);
|
|
||||||
d.id = count;
|
|
||||||
d.class = @intFromEnum(node.class);
|
|
||||||
const h = node.hid();
|
|
||||||
d.hid_len = @min(h.len, d.hid.len);
|
|
||||||
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
|
|
||||||
const rc = @min(node.resource_count, danos.max_dev_resources);
|
|
||||||
d.resource_count = rc;
|
|
||||||
for (0..rc) |i| {
|
|
||||||
const r = node.resources[i];
|
|
||||||
d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len };
|
|
||||||
}
|
|
||||||
devices[count] = d;
|
|
||||||
count += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
|
||||||
/// available (which may exceed `out.len`).
|
|
||||||
pub fn enumerate(out: []danos.DeviceDesc) usize {
|
|
||||||
const n = @min(count, out.len);
|
|
||||||
@memcpy(out[0..n], devices[0..n]);
|
|
||||||
return count;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
|
||||||
/// out of range or already claimed.
|
|
||||||
pub fn claim(id: u64, owner: u32) bool {
|
|
||||||
if (id >= count) return false;
|
|
||||||
if (claimed[@intCast(id)] != null) return false;
|
|
||||||
claimed[@intCast(id)] = owner;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The task that owns device `id`, or null.
|
|
||||||
pub fn ownerOf(id: u64) ?u32 {
|
|
||||||
if (id >= count) return null;
|
|
||||||
return claimed[@intCast(id)];
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Resource `idx` of device `id`, or null if out of range.
|
|
||||||
pub fn resourceOf(id: u64, idx: u64) ?danos.ResDesc {
|
|
||||||
if (id >= count) return null;
|
|
||||||
const d = &devices[@intCast(id)];
|
|
||||||
if (idx >= d.resource_count) return null;
|
|
||||||
return d.resources[@intCast(idx)];
|
|
||||||
}
|
|
||||||
+29
-29
@@ -2,7 +2,7 @@
|
|||||||
//!
|
//!
|
||||||
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
|
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
|
||||||
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
|
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
|
||||||
//! demand by mapping fresh frames into it (arch.mapPage) — the first real user of
|
//! demand by mapping fresh frames into it (architecture.mapPage) — the first real user of
|
||||||
//! the VMM (see docs/paging.md).
|
//! the VMM (see docs/paging.md).
|
||||||
//!
|
//!
|
||||||
//! The algorithm is a first-fit free list: an address-ordered singly linked list
|
//! The algorithm is a first-fit free list: an address-ordered singly linked list
|
||||||
@@ -14,17 +14,17 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const arch = @import("arch");
|
const architecture = @import("architecture");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
const page_size = danos.page_size;
|
||||||
|
|
||||||
/// Virtual base of the heap: the start of the higher half, which is unmapped and
|
/// Virtual base of the heap: the start of the higher half, which is unmapped and
|
||||||
/// well clear of the identity-mapped low half. (Canonical on x86_64; an arch that
|
/// well clear of the identity-mapped low half. (Canonical on x86_64; an architecture that
|
||||||
/// splits the address space differently would choose its own.)
|
/// splits the address space differently would choose its own.)
|
||||||
const heap_base: usize = 0xFFFF_8000_0000_0000;
|
const heap_base: usize = 0xFFFF_8000_0000_0000;
|
||||||
/// Cap on heap growth for now.
|
/// Cap on heap growth for now.
|
||||||
const heap_max: usize = 64 * 1024 * 1024;
|
const heap_maximum: usize = 64 * 1024 * 1024;
|
||||||
|
|
||||||
/// A block header, placed at the start of every block. While the block is free it
|
/// A block header, placed at the start of every block. While the block is free it
|
||||||
/// also links into the free list via `next`.
|
/// also links into the free list via `next`.
|
||||||
@@ -34,7 +34,7 @@ const Block = extern struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
const header_size = @sizeOf(Block); // 16
|
const header_size = @sizeOf(Block); // 16
|
||||||
const min_block = header_size + 16; // smallest block worth splitting off
|
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||||
|
|
||||||
var free_list: ?*Block = null;
|
var free_list: ?*Block = null;
|
||||||
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
|
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
|
||||||
@@ -56,15 +56,15 @@ pub fn init() void {
|
|||||||
|
|
||||||
/// Map more pages onto the end of the heap and add them as a free block. Returns
|
/// Map more pages onto the end of the heap and add them as a free block. Returns
|
||||||
/// false if out of heap virtual space or out of physical frames.
|
/// false if out of heap virtual space or out of physical frames.
|
||||||
fn grow(min_bytes: usize) bool {
|
fn grow(minimum_bytes: usize) bool {
|
||||||
const start = heap_end;
|
const start = heap_end;
|
||||||
const bytes = alignUp(min_bytes, page_size);
|
const bytes = alignUp(minimum_bytes, page_size);
|
||||||
if (start + bytes > heap_base + heap_max) return false;
|
if (start + bytes > heap_base + heap_maximum) return false;
|
||||||
|
|
||||||
var virt = start;
|
var virtual = start;
|
||||||
while (virt < start + bytes) : (virt += page_size) {
|
while (virtual < start + bytes) : (virtual += page_size) {
|
||||||
const frame = pmm.alloc() orelse return false;
|
const frame = pmm.alloc() orelse return false;
|
||||||
arch.mapPage(virt, frame, true);
|
architecture.mapPage(virtual, frame, true);
|
||||||
}
|
}
|
||||||
heap_end = start + bytes;
|
heap_end = start + bytes;
|
||||||
|
|
||||||
@@ -77,25 +77,25 @@ fn grow(min_bytes: usize) bool {
|
|||||||
/// Insert a block into the address-ordered free list, coalescing with the
|
/// Insert a block into the address-ordered free list, coalescing with the
|
||||||
/// physically adjacent free blocks on either side.
|
/// physically adjacent free blocks on either side.
|
||||||
fn insertFree(block: *Block) void {
|
fn insertFree(block: *Block) void {
|
||||||
var prev: ?*Block = null;
|
var previous: ?*Block = null;
|
||||||
var cur = free_list;
|
var current = free_list;
|
||||||
while (cur) |c| : (cur = c.next) {
|
while (current) |c| : (current = c.next) {
|
||||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||||
prev = c;
|
previous = c;
|
||||||
}
|
}
|
||||||
|
|
||||||
block.next = cur;
|
block.next = current;
|
||||||
if (prev) |p| p.next = block else free_list = block;
|
if (previous) |p| p.next = block else free_list = block;
|
||||||
|
|
||||||
// Merge forward into `cur` if they're contiguous.
|
// Merge forward into `current` if they're contiguous.
|
||||||
if (cur) |c| {
|
if (current) |c| {
|
||||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||||
block.size += c.size;
|
block.size += c.size;
|
||||||
block.next = c.next;
|
block.next = c.next;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Merge `prev` forward into `block` if they're contiguous.
|
// Merge `previous` forward into `block` if they're contiguous.
|
||||||
if (prev) |p| {
|
if (previous) |p| {
|
||||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||||
p.size += block.size;
|
p.size += block.size;
|
||||||
p.next = block.next;
|
p.next = block.next;
|
||||||
@@ -109,24 +109,24 @@ fn rawAlloc(len: usize) ?[*]u8 {
|
|||||||
|
|
||||||
var attempts: u32 = 0;
|
var attempts: u32 = 0;
|
||||||
while (attempts < 2) : (attempts += 1) {
|
while (attempts < 2) : (attempts += 1) {
|
||||||
var prev: ?*Block = null;
|
var previous: ?*Block = null;
|
||||||
var cur = free_list;
|
var current = free_list;
|
||||||
while (cur) |block| : ({
|
while (current) |block| : ({
|
||||||
prev = block;
|
previous = block;
|
||||||
cur = block.next;
|
current = block.next;
|
||||||
}) {
|
}) {
|
||||||
if (block.size < need) continue;
|
if (block.size < need) continue;
|
||||||
|
|
||||||
if (block.size >= need + min_block) {
|
if (block.size >= need + minimum_block) {
|
||||||
// Split: carve `need` off the front, leave the rest free.
|
// Split: carve `need` off the front, leave the rest free.
|
||||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||||
rest.size = block.size - need;
|
rest.size = block.size - need;
|
||||||
rest.next = block.next;
|
rest.next = block.next;
|
||||||
if (prev) |p| p.next = rest else free_list = rest;
|
if (previous) |p| p.next = rest else free_list = rest;
|
||||||
block.size = need;
|
block.size = need;
|
||||||
} else {
|
} else {
|
||||||
// Take the whole block.
|
// Take the whole block.
|
||||||
if (prev) |p| p.next = block.next else free_list = block.next;
|
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||||
}
|
}
|
||||||
return payloadOf(block);
|
return payloadOf(block);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,314 @@
|
|||||||
|
//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a
|
||||||
|
//! rendezvous point; a client `call`s it (send a message, block for a reply) and
|
||||||
|
//! a server `replyWait`s on it (reply to the last client, then block for the next
|
||||||
|
//! request). This is the substrate the user-space VFS server and device drivers
|
||||||
|
//! are reached through — `open`/`read`/`write` become user-space wrappers that
|
||||||
|
//! marshal a request into a `call`.
|
||||||
|
//!
|
||||||
|
//! Design (see docs/syscall.md, the plan):
|
||||||
|
//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame
|
||||||
|
//! through the physmap (`copyAcross`), which is mapped in every address space's
|
||||||
|
//! shared kernel half — so the kernel reads/writes either process's user memory
|
||||||
|
//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing.
|
||||||
|
//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply
|
||||||
|
//! to exactly one client at a time; that caller is held in `Task.ipc_client`.
|
||||||
|
//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without
|
||||||
|
//! becoming runnable*, which a WaitQueue can't express, so callers queue on the
|
||||||
|
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||||
|
//! waiting for work use a normal WaitQueue.
|
||||||
|
//!
|
||||||
|
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
||||||
|
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
||||||
|
//! copy is a later security-track item, matching the existing debug_write gap.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const danos = @import("danos");
|
||||||
|
const architecture = @import("architecture");
|
||||||
|
const scheduler = @import("scheduler.zig");
|
||||||
|
const sync = @import("sync.zig");
|
||||||
|
const heap = @import("heap.zig");
|
||||||
|
|
||||||
|
const page_size = danos.page_size;
|
||||||
|
const Task = scheduler.Task;
|
||||||
|
|
||||||
|
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
|
||||||
|
/// small because the copy runs under the big kernel lock.
|
||||||
|
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||||
|
|
||||||
|
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||||
|
pub const maximum_services = 8;
|
||||||
|
|
||||||
|
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||||
|
pub const EBADF: i64 = 1; // bad handle
|
||||||
|
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
||||||
|
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||||
|
pub const ENOENT: i64 = 4; // no such registered service
|
||||||
|
pub const ENOSPC: i64 = 5; // handle table or registry full
|
||||||
|
pub const ENOMEM: i64 = 6; // out of memory
|
||||||
|
|
||||||
|
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
||||||
|
/// message from a client — there is no reply owed. The low bits carry the source
|
||||||
|
/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in src/kernel/irq.zig;
|
||||||
|
/// the message path uses a plain task-id badge with this bit clear. Defined in the
|
||||||
|
/// shared contract (src/root.zig), because ring 3 has to test the same bit.
|
||||||
|
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||||
|
|
||||||
|
/// End of the user (low) canonical half — user buffers must lie below it.
|
||||||
|
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||||
|
|
||||||
|
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||||
|
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||||
|
pub const Endpoint = struct {
|
||||||
|
refcount: u32 = 1,
|
||||||
|
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||||
|
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||||
|
sender_head: ?*Task = null,
|
||||||
|
sender_tail: ?*Task = null,
|
||||||
|
// Servers blocked in `replyWait` awaiting a request.
|
||||||
|
receive_wait_queue: scheduler.WaitQueue = .{},
|
||||||
|
// Pending asynchronous notifications (badges), a small coalescing ring.
|
||||||
|
notify_buffer: [8]u64 = undefined,
|
||||||
|
notify_head: u8 = 0,
|
||||||
|
notify_tail: u8 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub fn createEndpoint() ?*Endpoint {
|
||||||
|
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||||
|
endpoint.* = .{};
|
||||||
|
return endpoint;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||||
|
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||||
|
pub fn dropRef(endpoint: *Endpoint) void {
|
||||||
|
if (endpoint.refcount > 1) {
|
||||||
|
endpoint.refcount -= 1;
|
||||||
|
} else {
|
||||||
|
heap.allocator().destroy(endpoint);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||||
|
|
||||||
|
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||||
|
t.next = null;
|
||||||
|
if (endpoint.sender_tail) |tail| tail.next = t else endpoint.sender_head = t;
|
||||||
|
endpoint.sender_tail = t;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dequeueSender(endpoint: *Endpoint) ?*Task {
|
||||||
|
const t = endpoint.sender_head orelse return null;
|
||||||
|
endpoint.sender_head = t.next;
|
||||||
|
if (endpoint.sender_head == null) endpoint.sender_tail = null;
|
||||||
|
t.next = null;
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- cross-address-space copy ----------------------------------------------
|
||||||
|
|
||||||
|
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
||||||
|
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||||
|
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||||
|
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
||||||
|
/// unmapped or out of range. Handles page-straddling buffers.
|
||||||
|
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
||||||
|
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
|
||||||
|
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
|
||||||
|
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
|
||||||
|
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
|
||||||
|
|
||||||
|
var off: usize = 0;
|
||||||
|
while (off < len) {
|
||||||
|
const s = architecture.translate(source_root, source_va + off) orelse return false;
|
||||||
|
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
|
||||||
|
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||||
|
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||||
|
const n = @min(@min(s_left, d_left), len - off);
|
||||||
|
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||||
|
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(d));
|
||||||
|
@memcpy(destination[0..n], source[0..n]);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
|
||||||
|
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
|
||||||
|
/// the range escapes the user half or any source page is unmapped — so a bad user
|
||||||
|
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
|
||||||
|
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
|
||||||
|
/// the machine). The correct way to pull a fixed-size struct in from user space, and
|
||||||
|
/// a single fetch: no TOCTOU against a hostile pointer.
|
||||||
|
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||||
|
if (user_as == 0) return false; // not a user address space
|
||||||
|
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
|
||||||
|
var off: usize = 0;
|
||||||
|
while (off < destination.len) {
|
||||||
|
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
||||||
|
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
||||||
|
const n = @min(s_left, destination.len - off);
|
||||||
|
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||||
|
@memcpy(destination[off..][0..n], source[0..n]);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the two IPC operations -------------------------------------------------
|
||||||
|
|
||||||
|
/// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a
|
||||||
|
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
|
||||||
|
/// negative errno. Runs as the current task.
|
||||||
|
pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
|
||||||
|
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
|
||||||
|
const me = scheduler.current();
|
||||||
|
me.ipc_send_ptr = message_ptr;
|
||||||
|
me.ipc_send_len = message_len;
|
||||||
|
me.ipc_reply_ptr = reply_ptr;
|
||||||
|
me.ipc_reply_cap = reply_cap;
|
||||||
|
me.ipc_status = 0;
|
||||||
|
|
||||||
|
enqueueSender(endpoint, me); // join the FIFO, then...
|
||||||
|
scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none)
|
||||||
|
scheduler.blockCurrentLocked(); // block until the reply readies us again
|
||||||
|
|
||||||
|
return me.ipc_status; // reply length or -errno, written by the replier
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client
|
||||||
|
/// we currently owe (if any), then receive the next request into
|
||||||
|
/// `[receive_ptr, receive_cap)`, blocking until one arrives. Writes the sender's badge
|
||||||
|
/// to `out_badge` and returns the request length, or a negative errno. A pending
|
||||||
|
/// notification is delivered ahead of client requests (length 0, badge with
|
||||||
|
/// `notify_badge_bit` set, no reply owed).
|
||||||
|
pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 {
|
||||||
|
if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
|
||||||
|
const me = scheduler.current();
|
||||||
|
|
||||||
|
// (1) Reply to the client we're still holding, if any.
|
||||||
|
if (me.ipc_client) |client| {
|
||||||
|
me.ipc_client = null;
|
||||||
|
const n = @min(reply_len, client.ipc_reply_cap);
|
||||||
|
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
||||||
|
client.ipc_status = @intCast(n);
|
||||||
|
} else {
|
||||||
|
client.ipc_status = -EFAULT;
|
||||||
|
}
|
||||||
|
scheduler.readyLocked(client); // its `call` now returns
|
||||||
|
}
|
||||||
|
|
||||||
|
// (2) Receive the next request (or notification), blocking until one is ready.
|
||||||
|
while (true) {
|
||||||
|
if (popNotify(endpoint)) |badge| {
|
||||||
|
out_badge.* = badge | notify_badge_bit;
|
||||||
|
return 0; // notification: no payload, no reply owed
|
||||||
|
}
|
||||||
|
if (dequeueSender(endpoint)) |caller| {
|
||||||
|
const n = @min(caller.ipc_send_len, receive_cap);
|
||||||
|
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
|
||||||
|
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||||
|
scheduler.readyLocked(caller);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
me.ipc_client = caller; // remember who to reply to
|
||||||
|
out_badge.* = caller.id;
|
||||||
|
return @intCast(n);
|
||||||
|
}
|
||||||
|
scheduler.waitLocked(&endpoint.receive_wait_queue); // nothing yet — sleep until woken, then retry
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- asynchronous notification (for IRQ-as-message, M10) --------------------
|
||||||
|
|
||||||
|
fn popNotify(endpoint: *Endpoint) ?u64 {
|
||||||
|
if (endpoint.notify_head == endpoint.notify_tail) return null;
|
||||||
|
const badge = endpoint.notify_buffer[endpoint.notify_head % endpoint.notify_buffer.len];
|
||||||
|
endpoint.notify_head +%= 1;
|
||||||
|
return badge;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post an asynchronous notification carrying `badge` to `endpoint` and wake a waiting
|
||||||
|
/// receiver. Precondition: the big kernel lock is held.
|
||||||
|
///
|
||||||
|
/// The lock must already cover whatever produced `endpoint` — an ISR that looked the
|
||||||
|
/// endpoint up in a table and *then* took the lock could be racing a process exit
|
||||||
|
/// that unbinds and frees it in between. See irq.dispatch, which holds one lock
|
||||||
|
/// region across the table read and this call.
|
||||||
|
///
|
||||||
|
/// A full ring drops the notification. That is the correct semantics, not a
|
||||||
|
/// concession: a notification is a *level* ("this device wants attention"), and the
|
||||||
|
/// driver re-reads device state on wake. It is never a count of events.
|
||||||
|
pub fn notifyLocked(endpoint: *Endpoint, badge: u64) void {
|
||||||
|
if (endpoint.notify_tail -% endpoint.notify_head < endpoint.notify_buffer.len) {
|
||||||
|
endpoint.notify_buffer[endpoint.notify_tail % endpoint.notify_buffer.len] = badge;
|
||||||
|
endpoint.notify_tail +%= 1;
|
||||||
|
}
|
||||||
|
scheduler.wakeLocked(&endpoint.receive_wait_queue);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `notifyLocked` as a self-contained ISR critical section, for a caller that holds
|
||||||
|
/// `endpoint` by some means other than a table the lock protects. Releases the lock without
|
||||||
|
/// touching the interrupt flag (the ISR's iretq restores it), like the timer tick.
|
||||||
|
pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
||||||
|
_ = sync.enter();
|
||||||
|
notifyLocked(endpoint, badge);
|
||||||
|
sync.leaveIsr();
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- per-process handle table + name registry -------------------------------
|
||||||
|
|
||||||
|
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
||||||
|
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
||||||
|
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||||
|
for (&t.handles, 0..) |*slot, i| {
|
||||||
|
if (slot.* == null) {
|
||||||
|
slot.* = @ptrCast(endpoint);
|
||||||
|
return @intCast(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -ENOSPC;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
||||||
|
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||||
|
if (h >= t.handles.len) return null;
|
||||||
|
const slot = t.handles[@intCast(h)] orelse return null;
|
||||||
|
return @ptrCast(@alignCast(slot));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
||||||
|
/// exit path so a dead server's endpoints don't linger referenced.
|
||||||
|
pub fn closeHandles(t: *Task) void {
|
||||||
|
for (&t.handles) |*slot| {
|
||||||
|
if (slot.*) |p| {
|
||||||
|
dropRef(@ptrCast(@alignCast(p)));
|
||||||
|
slot.* = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||||
|
|
||||||
|
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||||
|
pub fn register(id: u32, endpoint: *Endpoint) i64 {
|
||||||
|
if (id >= maximum_services) return -ENOENT;
|
||||||
|
if (registry[id]) |old| dropRef(old);
|
||||||
|
endpoint.refcount += 1;
|
||||||
|
registry[id] = endpoint;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find the endpoint published under `id`, taking a reference for the caller to
|
||||||
|
/// install in its handle table. Null if nothing is registered there.
|
||||||
|
pub fn lookup(id: u32) ?*Endpoint {
|
||||||
|
if (id >= maximum_services) return null;
|
||||||
|
const endpoint = registry[id] orelse return null;
|
||||||
|
endpoint.refcount += 1;
|
||||||
|
return endpoint;
|
||||||
|
}
|
||||||
+13
-13
@@ -4,14 +4,14 @@
|
|||||||
//! drivers and services live in separate address spaces, a message is how they
|
//! drivers and services live in separate address spaces, a message is how they
|
||||||
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
|
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
|
||||||
//! messages with a producer/consumer rendezvous, built on the scheduler's
|
//! messages with a producer/consumer rendezvous, built on the scheduler's
|
||||||
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `recv`
|
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `receive`
|
||||||
//! blocks when it's empty; neither busy-waits.
|
//! blocks when it's empty; neither busy-waits.
|
||||||
//!
|
//!
|
||||||
//! For now both endpoints are kernel threads sharing the kernel address space.
|
//! For now both endpoints are kernel threads sharing the kernel address space.
|
||||||
//! When user mode arrives, the same primitive carries messages across the
|
//! When user mode arrives, the same primitive carries messages across the
|
||||||
//! isolation boundary (with the payload copied between address spaces).
|
//! isolation boundary (with the payload copied between address spaces).
|
||||||
|
|
||||||
const sched = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
|
|
||||||
/// A bounded blocking channel of `capacity` messages of type `T`.
|
/// A bounded blocking channel of `capacity` messages of type `T`.
|
||||||
@@ -23,32 +23,32 @@ pub fn Channel(comptime T: type, comptime capacity: usize) type {
|
|||||||
head: usize = 0, // next slot to read
|
head: usize = 0, // next slot to read
|
||||||
tail: usize = 0, // next slot to write
|
tail: usize = 0, // next slot to write
|
||||||
count: usize = 0,
|
count: usize = 0,
|
||||||
not_full: sched.WaitQueue = .{}, // senders wait here
|
not_full: scheduler.WaitQueue = .{}, // senders wait here
|
||||||
not_empty: sched.WaitQueue = .{}, // receivers wait here
|
not_empty: scheduler.WaitQueue = .{}, // receivers wait here
|
||||||
|
|
||||||
/// Send a message, blocking while the channel is full.
|
/// Send a message, blocking while the channel is full.
|
||||||
pub fn send(self: *Self, msg: T) void {
|
pub fn send(self: *Self, message: T) void {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
// Recheck the condition in a loop: a wakeup only means "try again"
|
// Recheck the condition in a loop: a wakeup only means "try again"
|
||||||
// (another waiter may have taken the slot first).
|
// (another waiter may have taken the slot first).
|
||||||
while (self.count == capacity) sched.waitLocked(&self.not_full);
|
while (self.count == capacity) scheduler.waitLocked(&self.not_full);
|
||||||
self.buffer[self.tail] = msg;
|
self.buffer[self.tail] = message;
|
||||||
self.tail = (self.tail + 1) % capacity;
|
self.tail = (self.tail + 1) % capacity;
|
||||||
self.count += 1;
|
self.count += 1;
|
||||||
sched.wakeLocked(&self.not_empty); // a receiver can now proceed
|
scheduler.wakeLocked(&self.not_empty); // a receiver can now proceed
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Receive a message, blocking while the channel is empty.
|
/// Receive a message, blocking while the channel is empty.
|
||||||
pub fn recv(self: *Self) T {
|
pub fn receive(self: *Self) T {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
while (self.count == 0) sched.waitLocked(&self.not_empty);
|
while (self.count == 0) scheduler.waitLocked(&self.not_empty);
|
||||||
const msg = self.buffer[self.head];
|
const message = self.buffer[self.head];
|
||||||
self.head = (self.head + 1) % capacity;
|
self.head = (self.head + 1) % capacity;
|
||||||
self.count -= 1;
|
self.count -= 1;
|
||||||
sched.wakeLocked(&self.not_full); // a sender can now proceed
|
scheduler.wakeLocked(&self.not_full); // a sender can now proceed
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
return msg;
|
return message;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,277 +0,0 @@
|
|||||||
//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a
|
|
||||||
//! rendezvous point; a client `call`s it (send a message, block for a reply) and
|
|
||||||
//! a server `replyWait`s on it (reply to the last client, then block for the next
|
|
||||||
//! request). This is the substrate the user-space VFS server and device drivers
|
|
||||||
//! are reached through — `open`/`read`/`write` become user-space wrappers that
|
|
||||||
//! marshal a request into a `call`.
|
|
||||||
//!
|
|
||||||
//! Design (see docs/syscall.md, the plan):
|
|
||||||
//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame
|
|
||||||
//! through the physmap (`copyAcross`), which is mapped in every address space's
|
|
||||||
//! shared kernel half — so the kernel reads/writes either process's user memory
|
|
||||||
//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing.
|
|
||||||
//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply
|
|
||||||
//! to exactly one client at a time; that caller is held in `Task.ipc_client`.
|
|
||||||
//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without
|
|
||||||
//! becoming runnable*, which a WaitQueue can't express, so callers queue on the
|
|
||||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
|
||||||
//! waiting for work use a normal WaitQueue.
|
|
||||||
//!
|
|
||||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
|
||||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
|
||||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const sched = @import("scheduler.zig");
|
|
||||||
const sync = @import("sync.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
|
||||||
const Task = sched.Task;
|
|
||||||
|
|
||||||
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
|
|
||||||
/// small because the copy runs under the big kernel lock.
|
|
||||||
pub const MSG_MAX: usize = 256;
|
|
||||||
|
|
||||||
pub const max_handles = sched.ipc_max_handles;
|
|
||||||
pub const max_services = 8;
|
|
||||||
|
|
||||||
/// Errno-style failures, returned as `-value` in the syscall result register.
|
|
||||||
pub const EBADF: i64 = 1; // bad handle
|
|
||||||
pub const E2BIG: i64 = 2; // message exceeds MSG_MAX
|
|
||||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
|
||||||
pub const ENOENT: i64 = 4; // no such registered service
|
|
||||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
|
||||||
pub const ENOMEM: i64 = 6; // out of memory
|
|
||||||
|
|
||||||
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
|
||||||
/// message from a client — there is no reply owed. The low bits carry the source
|
|
||||||
/// (a GSI for IRQs). Used by `notifyFromIsr`/M10; the message path uses a plain
|
|
||||||
/// task-id badge with this bit clear.
|
|
||||||
pub const notify_badge_bit: u64 = 1 << 63;
|
|
||||||
|
|
||||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
|
||||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
|
||||||
|
|
||||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
|
||||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
|
||||||
pub const Endpoint = struct {
|
|
||||||
refcount: u32 = 1,
|
|
||||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
|
||||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
|
||||||
sender_head: ?*Task = null,
|
|
||||||
sender_tail: ?*Task = null,
|
|
||||||
// Servers blocked in `replyWait` awaiting a request.
|
|
||||||
recv_wq: sched.WaitQueue = .{},
|
|
||||||
// Pending asynchronous notifications (badges), a small coalescing ring.
|
|
||||||
notify_buf: [8]u64 = undefined,
|
|
||||||
notify_head: u8 = 0,
|
|
||||||
notify_tail: u8 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
pub fn createEndpoint() ?*Endpoint {
|
|
||||||
const ep = heap.allocator().create(Endpoint) catch return null;
|
|
||||||
ep.* = .{};
|
|
||||||
return ep;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
|
||||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
|
||||||
pub fn dropRef(ep: *Endpoint) void {
|
|
||||||
if (ep.refcount > 1) {
|
|
||||||
ep.refcount -= 1;
|
|
||||||
} else {
|
|
||||||
heap.allocator().destroy(ep);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
|
||||||
|
|
||||||
fn enqueueSender(ep: *Endpoint, t: *Task) void {
|
|
||||||
t.next = null;
|
|
||||||
if (ep.sender_tail) |tail| tail.next = t else ep.sender_head = t;
|
|
||||||
ep.sender_tail = t;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn dequeueSender(ep: *Endpoint) ?*Task {
|
|
||||||
const t = ep.sender_head orelse return null;
|
|
||||||
ep.sender_head = t.next;
|
|
||||||
if (ep.sender_head == null) ep.sender_tail = null;
|
|
||||||
t.next = null;
|
|
||||||
return t;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- cross-address-space copy ----------------------------------------------
|
|
||||||
|
|
||||||
/// Copy `len` bytes from `src_va` in address space `src_as` to `dst_va` in
|
|
||||||
/// `dst_as`, walking each side's page tables through the physmap (no CR3 switch).
|
|
||||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
|
||||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
|
||||||
/// unmapped or out of range. Handles page-straddling buffers.
|
|
||||||
fn copyAcross(src_as: u64, src_va: u64, dst_as: u64, dst_va: u64, len: usize) bool {
|
|
||||||
const src_root = if (src_as != 0) src_as else arch.kernelPageTable();
|
|
||||||
const dst_root = if (dst_as != 0) dst_as else arch.kernelPageTable();
|
|
||||||
if (src_as != 0 and (src_va >= user_half_end or src_va + len > user_half_end)) return false;
|
|
||||||
if (dst_as != 0 and (dst_va >= user_half_end or dst_va + len > user_half_end)) return false;
|
|
||||||
|
|
||||||
var off: usize = 0;
|
|
||||||
while (off < len) {
|
|
||||||
const s = arch.translate(src_root, src_va + off) orelse return false;
|
|
||||||
const d = arch.translate(dst_root, dst_va + off) orelse return false;
|
|
||||||
const s_left = page_size - ((src_va + off) & (page_size - 1));
|
|
||||||
const d_left = page_size - ((dst_va + off) & (page_size - 1));
|
|
||||||
const n = @min(@min(s_left, d_left), len - off);
|
|
||||||
const src: [*]const u8 = @ptrFromInt(danos.physToVirt(s));
|
|
||||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(d));
|
|
||||||
@memcpy(dst[0..n], src[0..n]);
|
|
||||||
off += n;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- the two IPC operations -------------------------------------------------
|
|
||||||
|
|
||||||
/// Client side of IPC_Call: send `[msg_ptr, msg_len)` to `ep` and block until a
|
|
||||||
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
|
|
||||||
/// negative errno. Runs as the current task.
|
|
||||||
pub fn call(ep: *Endpoint, msg_ptr: u64, msg_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
|
|
||||||
if (msg_len > MSG_MAX or reply_cap > MSG_MAX) return -E2BIG;
|
|
||||||
const flags = sync.enter();
|
|
||||||
defer sync.leave(flags);
|
|
||||||
|
|
||||||
const me = sched.cur();
|
|
||||||
me.ipc_send_ptr = msg_ptr;
|
|
||||||
me.ipc_send_len = msg_len;
|
|
||||||
me.ipc_reply_ptr = reply_ptr;
|
|
||||||
me.ipc_reply_cap = reply_cap;
|
|
||||||
me.ipc_status = 0;
|
|
||||||
|
|
||||||
enqueueSender(ep, me); // join the FIFO, then...
|
|
||||||
sched.wakeLocked(&ep.recv_wq); // ...wake a waiting server (no-op if none)
|
|
||||||
sched.blockCurrentLocked(); // block until the reply readies us again
|
|
||||||
|
|
||||||
return me.ipc_status; // reply length or -errno, written by the replier
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client
|
|
||||||
/// we currently owe (if any), then receive the next request into
|
|
||||||
/// `[recv_ptr, recv_cap)`, blocking until one arrives. Writes the sender's badge
|
|
||||||
/// to `out_badge` and returns the request length, or a negative errno. A pending
|
|
||||||
/// notification is delivered ahead of client requests (length 0, badge with
|
|
||||||
/// `notify_badge_bit` set, no reply owed).
|
|
||||||
pub fn replyWait(ep: *Endpoint, reply_ptr: u64, reply_len: u64, recv_ptr: u64, recv_cap: u64, out_badge: *u64) i64 {
|
|
||||||
if (reply_len > MSG_MAX or recv_cap > MSG_MAX) return -E2BIG;
|
|
||||||
const flags = sync.enter();
|
|
||||||
defer sync.leave(flags);
|
|
||||||
|
|
||||||
const me = sched.cur();
|
|
||||||
|
|
||||||
// (1) Reply to the client we're still holding, if any.
|
|
||||||
if (me.ipc_client) |client| {
|
|
||||||
me.ipc_client = null;
|
|
||||||
const n = @min(reply_len, client.ipc_reply_cap);
|
|
||||||
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
|
||||||
client.ipc_status = @intCast(n);
|
|
||||||
} else {
|
|
||||||
client.ipc_status = -EFAULT;
|
|
||||||
}
|
|
||||||
sched.readyLocked(client); // its `call` now returns
|
|
||||||
}
|
|
||||||
|
|
||||||
// (2) Receive the next request (or notification), blocking until one is ready.
|
|
||||||
while (true) {
|
|
||||||
if (popNotify(ep)) |badge| {
|
|
||||||
out_badge.* = badge | notify_badge_bit;
|
|
||||||
return 0; // notification: no payload, no reply owed
|
|
||||||
}
|
|
||||||
if (dequeueSender(ep)) |caller| {
|
|
||||||
const n = @min(caller.ipc_send_len, recv_cap);
|
|
||||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, recv_ptr, n)) {
|
|
||||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
|
||||||
sched.readyLocked(caller);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
me.ipc_client = caller; // remember who to reply to
|
|
||||||
out_badge.* = caller.id;
|
|
||||||
return @intCast(n);
|
|
||||||
}
|
|
||||||
sched.waitLocked(&ep.recv_wq); // nothing yet — sleep until woken, then retry
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- asynchronous notification (for IRQ-as-message, M10) --------------------
|
|
||||||
|
|
||||||
fn popNotify(ep: *Endpoint) ?u64 {
|
|
||||||
if (ep.notify_head == ep.notify_tail) return null;
|
|
||||||
const badge = ep.notify_buf[ep.notify_head % ep.notify_buf.len];
|
|
||||||
ep.notify_head +%= 1;
|
|
||||||
return badge;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Post an asynchronous notification carrying `badge` to `ep` and wake a waiting
|
|
||||||
/// receiver. ISR-safe: takes the big kernel lock and releases it without touching
|
|
||||||
/// the interrupt flag (the ISR's iretq restores it), exactly like the timer tick.
|
|
||||||
/// A full ring drops the notification (the driver re-reads device state anyway).
|
|
||||||
pub fn notifyFromIsr(ep: *Endpoint, badge: u64) void {
|
|
||||||
_ = sync.enter();
|
|
||||||
if (ep.notify_tail -% ep.notify_head < ep.notify_buf.len) {
|
|
||||||
ep.notify_buf[ep.notify_tail % ep.notify_buf.len] = badge;
|
|
||||||
ep.notify_tail +%= 1;
|
|
||||||
}
|
|
||||||
sched.wakeLocked(&ep.recv_wq);
|
|
||||||
sync.leaveIsr();
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- per-process handle table + name registry -------------------------------
|
|
||||||
|
|
||||||
/// Install `ep` in task `t`'s handle table; returns the small-int handle or
|
|
||||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
|
||||||
pub fn installHandle(t: *Task, ep: *Endpoint) i64 {
|
|
||||||
for (&t.handles, 0..) |*slot, i| {
|
|
||||||
if (slot.* == null) {
|
|
||||||
slot.* = @ptrCast(ep);
|
|
||||||
return @intCast(i);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return -ENOSPC;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
|
||||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
|
||||||
if (h >= t.handles.len) return null;
|
|
||||||
const slot = t.handles[@intCast(h)] orelse return null;
|
|
||||||
return @ptrCast(@alignCast(slot));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
|
||||||
/// exit path so a dead server's endpoints don't linger referenced.
|
|
||||||
pub fn closeHandles(t: *Task) void {
|
|
||||||
for (&t.handles) |*slot| {
|
|
||||||
if (slot.*) |p| {
|
|
||||||
dropRef(@ptrCast(@alignCast(p)));
|
|
||||||
slot.* = null;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
var registry: [max_services]?*Endpoint = .{null} ** max_services;
|
|
||||||
|
|
||||||
/// Publish `ep` under well-known `id` (takes a reference). Returns 0 or -errno.
|
|
||||||
pub fn register(id: u32, ep: *Endpoint) i64 {
|
|
||||||
if (id >= max_services) return -ENOENT;
|
|
||||||
if (registry[id]) |old| dropRef(old);
|
|
||||||
ep.refcount += 1;
|
|
||||||
registry[id] = ep;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Find the endpoint published under `id`, taking a reference for the caller to
|
|
||||||
/// install in its handle table. Null if nothing is registered there.
|
|
||||||
pub fn lookup(id: u32) ?*Endpoint {
|
|
||||||
if (id >= max_services) return null;
|
|
||||||
const ep = registry[id] orelse return null;
|
|
||||||
ep.refcount += 1;
|
|
||||||
return ep;
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,190 @@
|
|||||||
|
//! IRQ-as-IPC: delivering a hardware interrupt to a user-space driver.
|
||||||
|
//!
|
||||||
|
//! A microkernel can't run driver code in the ISR — the driver is a ring-3 process
|
||||||
|
//! in another address space. So the kernel's ISR does the least it can: quiet the
|
||||||
|
//! line, acknowledge the CPU, and post an asynchronous notification to the endpoint
|
||||||
|
//! the driver is blocked on (`ipc_sync.notifyFromIsr`). The driver wakes out of
|
||||||
|
//! `IPC_ReplyWait`, services the device, and calls `irq_ack` to re-arm.
|
||||||
|
//!
|
||||||
|
//! The full cycle, and why each step is where it is:
|
||||||
|
//!
|
||||||
|
//! ISR irqMask(gsi) -- the line is still asserted; stop it reaching a CPU
|
||||||
|
//! irqEoi() -- now safe to tell the LAPIC we're done
|
||||||
|
//! notifyFromIsr() -- wake the driver (it runs much later)
|
||||||
|
//! driver <services device> -- reads/clears the device's status register
|
||||||
|
//! driver irq_ack(device,resource) -- irqUnmask(gsi): the line is quiet, let it through
|
||||||
|
//!
|
||||||
|
//! Mask-before-EOI is the load-bearing part. A level-triggered line stays asserted
|
||||||
|
//! until the *device* is quieted, which only the ring-3 driver can do. EOI with the
|
||||||
|
//! entry unmasked and the I/O APIC redelivers immediately, forever, before the
|
||||||
|
//! driver is ever scheduled. Masking converts "level" into something a deferred
|
||||||
|
//! handler can cope with; `irq_ack` is what closes the loop.
|
||||||
|
//!
|
||||||
|
//! Binding is capability-gated exactly like `mmio_map`: the caller must have
|
||||||
|
//! `device_claim`ed the device, and the GSI must come from one of that device's `irq`
|
||||||
|
//! resources in the discovered device table (src/kernel/device-service.zig). A driver can
|
||||||
|
//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let
|
||||||
|
//! any process steal the keyboard's line.
|
||||||
|
//!
|
||||||
|
//! KNOWN ISSUE (real hardware, not QEMU). Masking a level-triggered redirection entry
|
||||||
|
//! while its remote-IRR bit is set does not clear remote-IRR on some chipsets, and the
|
||||||
|
//! line then never fires again — a driver would take exactly one interrupt and block
|
||||||
|
//! forever. QEMU's I/O APIC clears remote-IRR on EOI regardless of the mask, so the
|
||||||
|
//! `hpet` test cannot see this. Linux's workaround is to flush remote-IRR by briefly
|
||||||
|
//! flipping the entry to edge-trigger and back. Revisit when danos first boots on
|
||||||
|
//! metal; MSI (no mask cycle at all) sidesteps it entirely.
|
||||||
|
|
||||||
|
const architecture = @import("architecture");
|
||||||
|
const sync = @import("sync.zig");
|
||||||
|
const ipc_sync = @import("ipc-synchronous.zig");
|
||||||
|
|
||||||
|
/// GSIs a single I/O APIC covers. 24 is the standard redirection-table size; a
|
||||||
|
/// second I/O APIC (none on QEMU's q35) would extend this.
|
||||||
|
pub const maximum_gsi = 24;
|
||||||
|
|
||||||
|
/// The endpoint to notify for each bound GSI, or null if unbound. Read from the ISR
|
||||||
|
/// and written from syscalls, always under the big kernel lock.
|
||||||
|
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
|
||||||
|
|
||||||
|
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
|
||||||
|
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
|
||||||
|
/// out extra references), so "every GSI pointing at this endpoint" is not the same
|
||||||
|
/// set as "every GSI this process bound", and releasing the former on exit would mask
|
||||||
|
/// a live sibling's device line.
|
||||||
|
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
|
||||||
|
|
||||||
|
/// No GSI assigned to this vector.
|
||||||
|
const no_gsi: u32 = 0xFFFF_FFFF;
|
||||||
|
|
||||||
|
/// Reverse map for the ISR: which GSI does this vector carry? Populated at bind.
|
||||||
|
/// `interruptDispatch` hands a handler no arguments, so the vector→GSI edge has to
|
||||||
|
/// be recovered from somewhere — the trampolines below capture the vector at
|
||||||
|
/// comptime, and this turns it back into a GSI. Trampolines are installed on every
|
||||||
|
/// vector in the window at boot, so an unbound one must be distinguishable from
|
||||||
|
/// GSI 0 — hence the sentinel rather than a zero default.
|
||||||
|
var vector_gsi: [256]u32 = .{no_gsi} ** 256;
|
||||||
|
|
||||||
|
/// Vector currently assigned to each GSI (0 = none), so a rebind reuses it.
|
||||||
|
var gsi_vector: [maximum_gsi]u8 = .{0} ** maximum_gsi;
|
||||||
|
|
||||||
|
/// Set once the trampolines are installed.
|
||||||
|
var installed = false;
|
||||||
|
|
||||||
|
/// The ISR body for a bound device line. Runs with interrupts off, on the
|
||||||
|
/// interrupted task's kernel stack, on whichever core the I/O APIC picked.
|
||||||
|
fn dispatch(vector: u8) void {
|
||||||
|
const gsi = vector_gsi[vector];
|
||||||
|
if (gsi == no_gsi) {
|
||||||
|
// Nothing is routed here. Acknowledge so the LAPIC doesn't wedge on an
|
||||||
|
// in-service bit that never clears, but touch no redirection entry.
|
||||||
|
architecture.irqEoi();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// One lock region for the whole cycle. Two reasons, and the second is subtle:
|
||||||
|
//
|
||||||
|
// - The I/O APIC is an index/data register pair, so two cores interleaving a
|
||||||
|
// read-modify-write of a redirection entry would corrupt it.
|
||||||
|
// - `bound[gsi]` must be *read and used* under the same acquisition that
|
||||||
|
// `unbind` writes it under. Dropping the lock between the load and
|
||||||
|
// `notifyLocked` would let a driver exiting on another core free the endpoint
|
||||||
|
// in the gap, and we would post a notification into freed memory. The GSI is
|
||||||
|
// routed to the core that bound it, but a driver may migrate and exit
|
||||||
|
// elsewhere, so this is reachable on SMP.
|
||||||
|
//
|
||||||
|
// No deadlock: the lock is non-recursive, but a core holding it runs with
|
||||||
|
// interrupts disabled and so cannot interrupt itself into here.
|
||||||
|
_ = sync.enter();
|
||||||
|
defer sync.leaveIsr();
|
||||||
|
|
||||||
|
architecture.irqMask(gsi); // the line is still asserted; stop it reaching a CPU
|
||||||
|
architecture.irqEoi(); // now safe to release the LAPIC's in-service bit
|
||||||
|
|
||||||
|
// Wakes the driver if it's blocked in ReplyWait; otherwise queues the badge on
|
||||||
|
// the endpoint's notify ring, so an interrupt taken while the driver is off
|
||||||
|
// doing something else is not lost.
|
||||||
|
if (bound[gsi]) |endpoint| ipc_sync.notifyLocked(endpoint, gsi);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Install one no-argument trampoline per usable vector. Each closes over its own
|
||||||
|
/// `vector` as a comptime constant — that's the trick that gets an argument into
|
||||||
|
/// `idt.Handler` (`*const fn () void`) without a per-vector hand-written stub.
|
||||||
|
pub fn init() void {
|
||||||
|
if (installed) return;
|
||||||
|
inline for (0..architecture.irq_vector_count) |i| {
|
||||||
|
const vector: u8 = @intCast(@as(usize, architecture.irq_vector_base) + i);
|
||||||
|
architecture.irqSetHandler(vector, &struct {
|
||||||
|
fn trampoline() void {
|
||||||
|
dispatch(vector);
|
||||||
|
}
|
||||||
|
}.trampoline);
|
||||||
|
}
|
||||||
|
installed = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Lowest unused vector in the device window, or null if they're all spoken for.
|
||||||
|
fn allocVector() ?u8 {
|
||||||
|
var v: u8 = architecture.irq_vector_base;
|
||||||
|
while (v < architecture.irq_vector_base + architecture.irq_vector_count) : (v += 1) {
|
||||||
|
var used = false;
|
||||||
|
for (gsi_vector) |gv| {
|
||||||
|
if (gv == v) used = true;
|
||||||
|
}
|
||||||
|
if (!used) return v;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const BindError = error{ BadGsi, InUse, NoVector };
|
||||||
|
|
||||||
|
/// Deliver `gsi` to `endpoint` as an IPC notification, on behalf of task `owner`. Routes the
|
||||||
|
/// line to this core, installs the binding, and unmasks. Caller must hold the big
|
||||||
|
/// kernel lock, and must already have checked that `owner` claimed the device this GSI
|
||||||
|
/// belongs to.
|
||||||
|
pub fn bind(gsi: u32, endpoint: *ipc_sync.Endpoint, owner: u32) BindError!void {
|
||||||
|
if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi;
|
||||||
|
if (bound[gsi] != null) return error.InUse;
|
||||||
|
const vector = allocVector() orelse return error.NoVector;
|
||||||
|
|
||||||
|
vector_gsi[vector] = gsi;
|
||||||
|
gsi_vector[gsi] = vector;
|
||||||
|
bound[gsi] = endpoint;
|
||||||
|
bound_owner[gsi] = owner;
|
||||||
|
|
||||||
|
// Level-triggered, active-high. Level is the general case a driver must survive
|
||||||
|
// (and what hpetd configures its comparator for); an edge source simply never
|
||||||
|
// leaves the line asserted, so the mask/ack cycle is harmless there.
|
||||||
|
//
|
||||||
|
// Hardcoded for now: a device whose MADT interrupt-source override declares the
|
||||||
|
// line active-*low* (most legacy PCI INTx) will need the polarity threaded
|
||||||
|
// through from discovery. Nothing danos binds today is such a device.
|
||||||
|
architecture.irqRoute(gsi, vector, true, false);
|
||||||
|
architecture.irqUnmask(gsi);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-arm `gsi` after the driver has quieted the device. Caller holds the big lock
|
||||||
|
/// and has verified ownership.
|
||||||
|
pub fn ack(gsi: u32) bool {
|
||||||
|
if (gsi >= maximum_gsi or bound[gsi] == null) return false;
|
||||||
|
architecture.irqUnmask(gsi);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop every binding made by task `owner` — called as that task exits, *before* its
|
||||||
|
/// endpoints are freed. Each line is left masked, so a dead driver's device goes quiet
|
||||||
|
/// rather than interrupting into a freed endpoint. Caller holds the big kernel lock,
|
||||||
|
/// which is what makes this safe against a concurrent `dispatch` on another core.
|
||||||
|
///
|
||||||
|
/// Keyed on the owner, not the endpoint: endpoints are shared (a registered service's
|
||||||
|
/// endpoint has references in several processes), so releasing "everything pointing at
|
||||||
|
/// this endpoint" would tear down bindings this task never made.
|
||||||
|
pub fn releaseOwner(owner: u32) void {
|
||||||
|
for (&bound, 0..) |*slot, gsi| {
|
||||||
|
if (slot.* != null and bound_owner[gsi] == owner) {
|
||||||
|
architecture.irqMask(@intCast(gsi));
|
||||||
|
slot.* = null;
|
||||||
|
bound_owner[gsi] = 0;
|
||||||
|
gsi_vector[gsi] = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+11
-11
@@ -19,18 +19,18 @@
|
|||||||
//! and `recordPanic` (a breadcrumb in a fixed record).
|
//! and `recordPanic` (a breadcrumb in a fixed record).
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const arch = @import("arch");
|
const architecture = @import("architecture");
|
||||||
|
|
||||||
pub const SinkFn = *const fn ([]const u8) void;
|
pub const SinkFn = *const fn ([]const u8) void;
|
||||||
|
|
||||||
const max_sinks = 8;
|
const maximum_sinks = 8;
|
||||||
var sinks: [max_sinks]SinkFn = undefined;
|
var sinks: [maximum_sinks]SinkFn = undefined;
|
||||||
var sink_count: usize = 0;
|
var sink_count: usize = 0;
|
||||||
|
|
||||||
/// Register an output sink. Every registered sink receives every message; sinks
|
/// Register an output sink. Every registered sink receives every message; sinks
|
||||||
/// must be self-guarding (safe to call when their device is absent).
|
/// must be self-guarding (safe to call when their device is absent).
|
||||||
pub fn addSink(sink: SinkFn) void {
|
pub fn addSink(sink: SinkFn) void {
|
||||||
if (sink_count < max_sinks) {
|
if (sink_count < maximum_sinks) {
|
||||||
sinks[sink_count] = sink;
|
sinks[sink_count] = sink;
|
||||||
sink_count += 1;
|
sink_count += 1;
|
||||||
}
|
}
|
||||||
@@ -44,15 +44,15 @@ pub fn write(bytes: []const u8) void {
|
|||||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||||
/// this is safe to call from interrupt context and from a panic.
|
/// this is safe to call from interrupt context and from a panic.
|
||||||
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||||
var buf: [256]u8 = undefined;
|
var buffer: [256]u8 = undefined;
|
||||||
write(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
write(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
|
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
|
||||||
/// progress channel for when there is no text output at all. Independent of the
|
/// progress channel for when there is no text output at all. Independent of the
|
||||||
/// sink list, so it works even before any sink is registered.
|
/// sink list, so it works even before any sink is registered.
|
||||||
pub fn checkpoint(code: u8) void {
|
pub fn checkpoint(code: u8) void {
|
||||||
arch.checkpoint(code);
|
architecture.checkpoint(code);
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- persistent panic breadcrumb -------------------------------------------
|
// --- persistent panic breadcrumb -------------------------------------------
|
||||||
@@ -68,16 +68,16 @@ pub const PanicRecord = extern struct {
|
|||||||
magic: u64 = 0,
|
magic: u64 = 0,
|
||||||
len: u32 = 0,
|
len: u32 = 0,
|
||||||
_pad: u32 = 0,
|
_pad: u32 = 0,
|
||||||
msg: [512]u8 = undefined,
|
message: [512]u8 = undefined,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Findable by symbol (`log.panic_record`) for a debugger or RAM dump.
|
/// Findable by symbol (`log.panic_record`) for a debugger or RAM dump.
|
||||||
pub var panic_record: PanicRecord = .{};
|
pub var panic_record: PanicRecord = .{};
|
||||||
|
|
||||||
/// Stamp the panic message into the breadcrumb record.
|
/// Stamp the panic message into the breadcrumb record.
|
||||||
pub fn recordPanic(msg: []const u8) void {
|
pub fn recordPanic(message: []const u8) void {
|
||||||
const n: u32 = @intCast(@min(msg.len, panic_record.msg.len));
|
const n: u32 = @intCast(@min(message.len, panic_record.message.len));
|
||||||
@memcpy(panic_record.msg[0..n], msg[0..n]);
|
@memcpy(panic_record.message[0..n], message[0..n]);
|
||||||
panic_record.len = n;
|
panic_record.len = n;
|
||||||
panic_record.magic = panic_magic; // set last: a reader sees a complete record
|
panic_record.magic = panic_magic; // set last: a reader sees a complete record
|
||||||
}
|
}
|
||||||
|
|||||||
+94
-85
@@ -1,24 +1,25 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
const arch = @import("arch");
|
const architecture = @import("architecture");
|
||||||
const console = @import("console.zig");
|
const console = @import("console.zig");
|
||||||
const log = @import("log.zig");
|
const log = @import("log.zig");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const process = @import("process.zig");
|
const process = @import("process.zig");
|
||||||
const devsvc = @import("devsvc.zig");
|
const device_service = @import("device-service.zig");
|
||||||
|
const irq = @import("irq.zig");
|
||||||
const initrd = @import("initrd");
|
const initrd = @import("initrd");
|
||||||
const platform = @import("platform");
|
const platform = @import("platform");
|
||||||
const tests = @import("tests.zig");
|
const tests = @import("tests.zig");
|
||||||
const build_options = @import("build_options");
|
const build_options = @import("build_options");
|
||||||
const BootInfo = danos.BootInfo;
|
const BootInformation = danos.BootInformation;
|
||||||
|
|
||||||
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
|
/// The calling convention used to enter the kernel. Pinned to SystemV explicitly:
|
||||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
||||||
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
|
/// x64 (first argument in RCX), while the kernel is SystemV (first argument in
|
||||||
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
|
/// RDI). Both sides reference this so the `boot_information` pointer lands in the
|
||||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
||||||
pub const kernel_abi = danos.kernel_abi;
|
pub const kernel_abi = danos.kernel_abi;
|
||||||
|
|
||||||
@@ -43,38 +44,38 @@ var ap_trampoline_page: u64 = 0;
|
|||||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||||
/// caller to return to, so this never returns.
|
/// caller to return to, so this never returns.
|
||||||
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
|
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
|
||||||
/// then calls this with the loader's `boot_info` pointer in RDI. We can't keep
|
/// then calls this with the loader's `boot_information` pointer in RDI. We can't keep
|
||||||
/// running on the loader's stack: it's a low physical address that the identity
|
/// running on the loader's stack: it's a low physical address that the identity
|
||||||
/// map covers only transitionally, and vanishes once the kernel drops the low
|
/// map covers only transitionally, and vanishes once the kernel drops the low
|
||||||
/// half. `boot_info` (also low) is reached through the physmap — its base is the
|
/// half. `boot_information` (also low) is reached through the physmap — its base is the
|
||||||
/// same under the loader's bootstrap tables and the kernel's own.
|
/// same under the loader's bootstrap tables and the kernel's own.
|
||||||
export fn kmainEntry(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
export fn kmainEntry(boot_information: *const BootInformation) callconv(kernel_abi) noreturn {
|
||||||
kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info))));
|
kmain(@ptrFromInt(danos.physicalToVirtual(@intFromPtr(boot_information))));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
fn kmain(boot_information: *const BootInformation) noreturn {
|
||||||
// The **log** is the machine-readable diagnostic stream: it fans out to every
|
// The **log** is the machine-readable diagnostic stream: it fans out to every
|
||||||
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
|
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
|
||||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||||
// with port-0x80 checkpoints as the only progress signal.
|
// with port-0x80 checkpoints as the only progress signal.
|
||||||
arch.serialInit();
|
architecture.serialInit();
|
||||||
log.addSink(arch.serialWrite);
|
log.addSink(architecture.serialWrite);
|
||||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite);
|
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||||
|
|
||||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||||
const fb = boot_info.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
console.init(fb);
|
console.init(fb);
|
||||||
|
|
||||||
log.checkpoint(cp_entry);
|
log.checkpoint(cp_entry);
|
||||||
|
|
||||||
// Catch CPU exceptions before doing anything that might fault: install our
|
// Catch CPU exceptions before doing anything that might fault: install our
|
||||||
// reporter, then bring up the GDT + IDT.
|
// reporter, then bring up the GDT + IDT.
|
||||||
arch.setFaultHandler(onException);
|
architecture.setFaultHandler(onException);
|
||||||
arch.init();
|
architecture.init();
|
||||||
|
|
||||||
status("danos: initialising kernel...\n");
|
status("danos: initialising kernel...\n");
|
||||||
log.write(if (console.present())
|
log.write(if (console.present())
|
||||||
@@ -90,7 +91,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
|
|
||||||
// Summarise the physical memory the loader handed us. The array is danos's
|
// Summarise the physical memory the loader handed us. The array is danos's
|
||||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(boot_info.memory_map.regions)))[0..boot_info.memory_map.len];
|
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||||
var usable_pages: u64 = 0;
|
var usable_pages: u64 = 0;
|
||||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||||
for (regions) |r| {
|
for (regions) |r| {
|
||||||
@@ -112,7 +113,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
|
|
||||||
// Bring up the physical frame allocator over that map, and prove it works:
|
// Bring up the physical frame allocator over that map, and prove it works:
|
||||||
// allocate three frames, then hand them back.
|
// allocate three frames, then hand them back.
|
||||||
pmm.init(boot_info.memory_map);
|
pmm.init(boot_information.memory_map);
|
||||||
// Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap
|
// Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap
|
||||||
// draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held
|
// draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held
|
||||||
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
||||||
@@ -130,11 +131,11 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||||
|
|
||||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||||
arch.enablePaging(pmm.alloc, pmm.free, boot_info);
|
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
||||||
log.checkpoint(cp_paging);
|
log.checkpoint(cp_paging);
|
||||||
log.print("\ndanos: paging enabled\n", .{});
|
log.print("\ndanos: paging enabled\n", .{});
|
||||||
log.print(" page tables: root = 0x{x:0>16}\n", .{arch.activePageTable()});
|
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||||
heap.init();
|
heap.init();
|
||||||
@@ -146,24 +147,32 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
|
|
||||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
||||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
||||||
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
// mapped) and maps PCIe configuration space on demand via the VMM. A failure here is
|
||||||
// not fatal yet — log it and carry on.
|
// not fatal yet — log it and carry on.
|
||||||
const hal = platform.Hal{
|
const hal = platform.Hal{
|
||||||
.mapMmio = arch.mapMmio,
|
.mapMmio = architecture.mapMmio,
|
||||||
.pioRead = arch.pioRead,
|
.pioRead = architecture.pioRead,
|
||||||
.pioWrite = arch.pioWrite,
|
.pioWrite = architecture.pioWrite,
|
||||||
};
|
};
|
||||||
if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| {
|
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
||||||
var dt = devtree;
|
var device_tree = devtree;
|
||||||
log.write("\ndanos: device discovery online\n");
|
log.write("\ndanos: device discovery online\n");
|
||||||
dt.dump(log.write);
|
device_tree.dump(log.write);
|
||||||
|
|
||||||
// Snapshot the device tree for user-space drivers (dev_enumerate/claim/
|
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||||
// mmio_map operate on this flat, id-indexed table + claim map).
|
// mmio_map operate on this flat, id-indexed table + claim map).
|
||||||
devsvc.init(&dt);
|
device_service.init(&device_tree);
|
||||||
|
if (device_service.dropped > 0) {
|
||||||
|
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||||
|
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{device_service.dropped});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||||
|
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||||
|
irq.init();
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||||
const pw = platform.powerInfo();
|
const pw = platform.powerInformation();
|
||||||
log.write("danos: power\n");
|
log.write("danos: power\n");
|
||||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||||
if (pw.s5) |s| {
|
if (pw.s5) |s| {
|
||||||
@@ -177,32 +186,32 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
const am = platform.amlStats();
|
const am = platform.amlStats();
|
||||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||||
|
|
||||||
// Feed the arch layer the discovered addresses/facts so it makes no legacy
|
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||||
const pinfo = platform.platformInfo();
|
const pinfo = platform.platformInformation();
|
||||||
const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t|
|
const hpet_base: u64 = if (device_tree.firstOfClass(.timer)) |t|
|
||||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
(if (t.firstResource(.memory)) |r| r.start else 0)
|
||||||
else
|
else
|
||||||
0;
|
0;
|
||||||
var ioapic_base: u64 = 0;
|
var ioapic_base: u64 = 0;
|
||||||
var ioapic_gsi: u32 = 0;
|
var ioapic_gsi: u32 = 0;
|
||||||
if (dt.firstOfClass(.interrupt_controller)) |ic| {
|
if (device_tree.firstOfClass(.interrupt_controller)) |ic| {
|
||||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
||||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
||||||
}
|
}
|
||||||
var isos: [16]arch.IsoEntry = undefined;
|
var isos: [16]architecture.IsoEntry = undefined;
|
||||||
const iso_n = @min(pinfo.override_count, isos.len);
|
const iso_n = @min(pinfo.override_count, isos.len);
|
||||||
for (0..iso_n) |i| isos[i] = .{
|
for (0..iso_n) |i| isos[i] = .{
|
||||||
.source = pinfo.overrides[i].source,
|
.source = pinfo.overrides[i].source,
|
||||||
.gsi = pinfo.overrides[i].gsi,
|
.gsi = pinfo.overrides[i].gsi,
|
||||||
.flags = pinfo.overrides[i].flags,
|
.flags = pinfo.overrides[i].flags,
|
||||||
};
|
};
|
||||||
const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present())
|
const pm_timer: ?architecture.PmTimer = if (pinfo.pm_timer.present())
|
||||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
||||||
else
|
else
|
||||||
null;
|
null;
|
||||||
arch.configurePlatform(.{
|
architecture.configurePlatform(.{
|
||||||
.pic_present = pinfo.pic_present,
|
.pic_present = pinfo.pic_present,
|
||||||
.hpet_base = hpet_base,
|
.hpet_base = hpet_base,
|
||||||
.pm_timer = pm_timer,
|
.pm_timer = pm_timer,
|
||||||
@@ -210,7 +219,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
.ioapic_gsi_base = ioapic_gsi,
|
.ioapic_gsi_base = ioapic_gsi,
|
||||||
.overrides = isos[0..iso_n],
|
.overrides = isos[0..iso_n],
|
||||||
});
|
});
|
||||||
if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address);
|
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
||||||
|
|
||||||
log.write("danos: platform\n");
|
log.write("danos: platform\n");
|
||||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||||
@@ -222,7 +231,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
} else {
|
} else {
|
||||||
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
||||||
}
|
}
|
||||||
log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, arch.irqRouteCount(), arch.irqRouteRaw(0) });
|
log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, architecture.irqRouteCount(), architecture.irqRouteRaw(0) });
|
||||||
const cores = platform.cpus();
|
const cores = platform.cpus();
|
||||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||||
if (platform.cpusDropped() > 0)
|
if (platform.cpusDropped() > 0)
|
||||||
@@ -232,7 +241,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
}
|
}
|
||||||
log.checkpoint(cp_discovery);
|
log.checkpoint(cp_discovery);
|
||||||
|
|
||||||
// Install the syscall handler (int 0x80 gate + syscall stub) once, before any
|
// Install the system_call handler (int 0x80 gate + system_call stub) once, before any
|
||||||
// user code runs.
|
// user code runs.
|
||||||
process.init();
|
process.init();
|
||||||
|
|
||||||
@@ -243,10 +252,10 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
|
|
||||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||||
// the timer preempts among tasks.
|
// the timer preempts among tasks.
|
||||||
arch.startTimer();
|
architecture.startTimer();
|
||||||
arch.enableInterrupts();
|
architecture.enableInterrupts();
|
||||||
log.checkpoint(cp_timer);
|
log.checkpoint(cp_timer);
|
||||||
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.timerClockHz() / 1_000_000, arch.clockHz() / 1_000_000, arch.timerCalibrationSource() });
|
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||||
|
|
||||||
// Wake the other cores (application processors). A no-op on a single-core
|
// Wake the other cores (application processors). A no-op on a single-core
|
||||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
||||||
@@ -255,8 +264,8 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||||
// Normal builds fall through to the idle halt.
|
// Normal builds fall through to the idle halt.
|
||||||
if (build_options.test_case) |case| {
|
if (build_options.test_case) |case| {
|
||||||
tests.run(case, boot_info);
|
tests.run(case, boot_information);
|
||||||
arch.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
log.checkpoint(cp_running);
|
log.checkpoint(cp_running);
|
||||||
@@ -266,9 +275,9 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
// loader) and spawn it as a real ring-3 process, PID 1. It runs on its own
|
// loader) and spawn it as a real ring-3 process, PID 1. It runs on its own
|
||||||
// address space, preemptively, alongside the kernel — no cooperative
|
// address space, preemptively, alongside the kernel — no cooperative
|
||||||
// borrowing. This boot context then becomes the BSP's idle loop.
|
// borrowing. This boot context then becomes the BSP's idle loop.
|
||||||
if (boot_info.init_len != 0) {
|
if (boot_information.init_len != 0) {
|
||||||
status("starting /sbin/init...\n");
|
status("starting /sbin/init...\n");
|
||||||
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||||
process.spawnProcess(image, 4) catch |err| {
|
process.spawnProcess(image, 4) catch |err| {
|
||||||
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
||||||
};
|
};
|
||||||
@@ -278,22 +287,22 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
|
|
||||||
// Spawn the extra user binaries the loader ferried in the initrd (the VFS
|
// Spawn the extra user binaries the loader ferried in the initrd (the VFS
|
||||||
// server, and later device drivers). For now the kernel launches them all;
|
// server, and later device drivers). For now the kernel launches them all;
|
||||||
// once init is a real service supervisor it will spawn them itself (sys_spawn).
|
// once init is a real service supervisor it will spawn them itself (system_spawn).
|
||||||
startInitrdBinaries(boot_info);
|
startInitrdBinaries(boot_information);
|
||||||
|
|
||||||
// Become the idle task: drop below every real task and halt until an
|
// Become the idle task: drop below every real task and halt until an
|
||||||
// interrupt. The timer keeps preempting into init and any other work.
|
// interrupt. The timer keeps preempting into init and any other work.
|
||||||
scheduler.setPriority(0);
|
scheduler.setPriority(0);
|
||||||
status("\nkernel idle; /sbin/init is running.\n");
|
status("\nkernel idle; /sbin/init is running.\n");
|
||||||
arch.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn every program bundled in the initrd as its own ring-3 process. A bad
|
/// Spawn every program bundled in the initrd as its own ring-3 process. A bad
|
||||||
/// image or a program that fails to load is logged and skipped — the rest of the
|
/// image or a program that fails to load is logged and skipped — the rest of the
|
||||||
/// system still runs.
|
/// system still runs.
|
||||||
fn startInitrdBinaries(boot_info: *const danos.BootInfo) void {
|
fn startInitrdBinaries(boot_information: *const danos.BootInformation) void {
|
||||||
if (boot_info.initrd_len == 0) return;
|
if (boot_information.initrd_len == 0) return;
|
||||||
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.initrd_base)))[0..boot_info.initrd_len];
|
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||||
const rd = initrd.Reader.init(image) orelse {
|
const rd = initrd.Reader.init(image) orelse {
|
||||||
status("initrd: bad image, skipping\n");
|
status("initrd: bad image, skipping\n");
|
||||||
return;
|
return;
|
||||||
@@ -323,40 +332,40 @@ fn bringUpSecondaries() void {
|
|||||||
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
arch.setTrampolinePage(ap_trampoline_page);
|
architecture.setTrampolinePage(ap_trampoline_page);
|
||||||
arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
|
architecture.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
|
||||||
|
|
||||||
// Test hook: the smp-retry case forces the first wake to fail, so the retry below
|
// Test hook: the smp-retry case forces the first wake to fail, so the retry below
|
||||||
// must still bring every core online. Inert in a normal build (test_case is null).
|
// must still bring every core online. Inert in a normal build (test_case is null).
|
||||||
if (build_options.test_case) |tc| {
|
if (build_options.test_case) |tc| {
|
||||||
if (std.mem.eql(u8, tc, "smp-retry")) arch.testFailNextWakes(1);
|
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||||
const max_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
||||||
for (cores[1..], 1..) |core, index| {
|
for (cores[1..], 1..) |core, index| {
|
||||||
const stack = heap.allocator().alloc(u8, config.kernel_stack_size) catch {
|
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
||||||
log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id});
|
log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id});
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
||||||
// This core's dedicated fault stack — allocated only now that the core is
|
// This core's dedicated fault stack — allocated only now that the core is
|
||||||
// real, rather than reserved statically for every possible core.
|
// real, rather than reserved statically for every possible core.
|
||||||
const fault_stack = heap.allocator().alloc(u8, arch.fault_stack_size) catch {
|
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
||||||
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
arch.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||||
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
||||||
var attempt: u32 = 1;
|
var attempt: u32 = 1;
|
||||||
while (attempt <= max_wake_attempts) : (attempt += 1) {
|
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
||||||
if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||||
pc.online = true;
|
pc.online = true;
|
||||||
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (attempt == max_wake_attempts)
|
if (attempt == maximum_wake_attempts)
|
||||||
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, max_wake_attempts });
|
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||||
@@ -365,14 +374,14 @@ fn bringUpSecondaries() void {
|
|||||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
||||||
/// touches the framebuffer.
|
/// touches the framebuffer.
|
||||||
fn status(msg: []const u8) void {
|
fn status(message: []const u8) void {
|
||||||
log.write(msg);
|
log.write(message);
|
||||||
console.write(msg);
|
console.write(message);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
||||||
var buf: [256]u8 = undefined;
|
var buffer: [256]u8 = undefined;
|
||||||
status(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Frames (4 KiB pages) to whole MiB.
|
/// Frames (4 KiB pages) to whole MiB.
|
||||||
@@ -390,33 +399,33 @@ fn kib(frames: u64) u64 {
|
|||||||
/// running (full recovery — kill the task, keep the core — is the resilience track,
|
/// running (full recovery — kill the task, keep the core — is the resilience track,
|
||||||
/// see docs/resilience.md). The report names the core so an AP fault is attributed,
|
/// see docs/resilience.md). The report names the core so an AP fault is attributed,
|
||||||
/// and goes to every output sink plus a POST code and a persistent breadcrumb.
|
/// and goes to every output sink plus a POST code and a persistent breadcrumb.
|
||||||
fn onException(state: *const arch.CpuState) noreturn {
|
fn onException(state: *const architecture.CpuState) noreturn {
|
||||||
log.checkpoint(cp_exception);
|
log.checkpoint(cp_exception);
|
||||||
const core = scheduler.currentCpuIndex();
|
const core = scheduler.currentCpuIndex();
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||||
// top of the diagnostic log.
|
// top of the diagnostic log.
|
||||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.exceptionName(state.vector), state.vector });
|
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
statusPrint(" IP : 0x{x:0>16}\n", .{arch.instructionPointer(state)});
|
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||||
statusPrint(" SP : 0x{x:0>16}\n", .{arch.stackPointer(state)});
|
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||||
if (arch.faultAddress(state)) |addr| statusPrint(" fault addr : 0x{x:0>16}\n", .{addr});
|
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||||
|
|
||||||
var buf: [128]u8 = undefined;
|
var buffer: [128]u8 = undefined;
|
||||||
log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ arch.exceptionName(state.vector), state.vector, core, arch.instructionPointer(state) }) catch "cpu exception");
|
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||||
arch.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
||||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
||||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
||||||
pub const panic = std.debug.FullPanic(struct {
|
pub const panic = std.debug.FullPanic(struct {
|
||||||
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
|
fn panic(message: []const u8, first_trace_address: ?usize) noreturn {
|
||||||
_ = first_trace_addr;
|
_ = first_trace_address;
|
||||||
log.checkpoint(cp_panic);
|
log.checkpoint(cp_panic);
|
||||||
log.recordPanic(msg);
|
log.recordPanic(message);
|
||||||
status("\nKERNEL PANIC: ");
|
status("\nKERNEL PANIC: ");
|
||||||
status(msg);
|
status(message);
|
||||||
status("\n");
|
status("\n");
|
||||||
arch.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
}.panic);
|
}.panic);
|
||||||
|
|||||||
+4
-4
@@ -49,7 +49,7 @@ inline fn setFree(frame: usize) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(map.regions)))[0..map.len];
|
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(map.regions)))[0..map.len];
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the allocator from the loader's memory map. Reaches physical memory
|
/// Build the allocator from the loader's memory map. Reaches physical memory
|
||||||
@@ -91,7 +91,7 @@ pub fn init(map: danos.MemoryMap) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
||||||
bitmap = @as([*]u8, @ptrFromInt(danos.physToVirt(bitmap_base)))[0..bitmap_bytes];
|
bitmap = @as([*]u8, @ptrFromInt(danos.physicalToVirtual(bitmap_base)))[0..bitmap_bytes];
|
||||||
|
|
||||||
// 3. Start with everything marked used, then free the usable regions. Doing
|
// 3. Start with everything marked used, then free the usable regions. Doing
|
||||||
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
||||||
@@ -165,8 +165,8 @@ pub fn allocBelow(limit: u64) ?u64 {
|
|||||||
|
|
||||||
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
|
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
|
||||||
/// ignored rather than corrupting the count.
|
/// ignored rather than corrupting the count.
|
||||||
pub fn free(addr: u64) void {
|
pub fn free(address: u64) void {
|
||||||
const f: usize = @intCast(addr / page_size);
|
const f: usize = @intCast(address / page_size);
|
||||||
if (f >= total_frames or !isUsed(f)) return;
|
if (f >= total_frames or !isUsed(f)) return;
|
||||||
setFree(f);
|
setFree(f);
|
||||||
used_frames -= 1;
|
used_frames -= 1;
|
||||||
|
|||||||
+243
-158
@@ -10,7 +10,7 @@
|
|||||||
//! *current* kernel context via the borrowed-thread path — a minimal probe of
|
//! *current* kernel context via the borrowed-thread path — a minimal probe of
|
||||||
//! the ring-transition mechanisms, kept for that test.
|
//! the ring-transition mechanisms, kept for that test.
|
||||||
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
|
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
|
||||||
//! talks to the kernel only through the syscall instruction (or the int 0x80
|
//! talks to the kernel only through the system_call instruction (or the int 0x80
|
||||||
//! gate). The shared handler is installed once by `init`.
|
//! gate). The shared handler is installed once by `init`.
|
||||||
//!
|
//!
|
||||||
//! Borrowed-path caveat (`run` only): it publishes TSS.rsp0 on the *current*
|
//! Borrowed-path caveat (`run` only): it publishes TSS.rsp0 on the *current*
|
||||||
@@ -22,23 +22,24 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const danos = @import("danos");
|
const danos = @import("danos");
|
||||||
const arch = @import("arch");
|
const architecture = @import("architecture");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const sched = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const ipc = @import("ipc_sync.zig");
|
const ipc = @import("ipc-synchronous.zig");
|
||||||
const devsvc = @import("devsvc.zig");
|
const device_service = @import("device-service.zig");
|
||||||
|
const irq = @import("irq.zig");
|
||||||
const log = @import("log.zig");
|
const log = @import("log.zig");
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
const page_size = danos.page_size;
|
||||||
const Syscall = danos.Syscall;
|
const SystemCall = danos.SystemCall;
|
||||||
|
|
||||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||||
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
||||||
/// An ELF image may occupy [code_virt, stack_virt); the stack page sits above.
|
/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above.
|
||||||
pub const code_virt: u64 = 0x0000_7000_0000_0000;
|
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
|
||||||
pub const stack_virt: u64 = 0x0000_7000_0020_0000;
|
pub const stack_virtual: u64 = 0x0000_7000_0020_0000;
|
||||||
|
|
||||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||||
@@ -48,19 +49,19 @@ pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
|||||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||||
|
|
||||||
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
|
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
|
||||||
/// used to bound the addresses a syscall will dereference on the caller's behalf.
|
/// used to bound the addresses a system_call will dereference on the caller's behalf.
|
||||||
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||||
|
|
||||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||||
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
||||||
/// `Task.dev_map_next`.
|
/// `Task.device_map_next`.
|
||||||
pub const dev_arena_base: u64 = 0x0000_7100_0000_0000;
|
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||||
pub const dev_arena_end: u64 = dev_arena_base + (4 << 30);
|
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||||
|
|
||||||
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
||||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
||||||
const max_mmap_pages = 256;
|
const maximum_mmap_pages = 256;
|
||||||
|
|
||||||
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
||||||
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
||||||
@@ -71,167 +72,251 @@ pub fn pfBlob() []const u8 {
|
|||||||
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
|
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
|
||||||
}
|
}
|
||||||
|
|
||||||
/// What debug_write syscalls produced (accumulated), and the exit syscall's code.
|
/// What debug_write syscalls produced (accumulated), and the exit system_call's code.
|
||||||
pub var write_buf: [256]u8 = undefined;
|
pub var write_buffer: [256]u8 = undefined;
|
||||||
pub var write_len: usize = 0;
|
pub var write_len: usize = 0;
|
||||||
pub var write_from_user: bool = false;
|
pub var write_from_user: bool = false;
|
||||||
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
||||||
pub var exit_code: u64 = 0;
|
pub var exit_code: u64 = 0;
|
||||||
|
|
||||||
/// The syscall surface, dispatched on the saved syscall number (`danos.Syscall`).
|
/// The system_call surface, dispatched on the saved system_call number (`danos.SystemCall`).
|
||||||
/// This is the microkernel-minimal set — memory + scheduling only; file/device
|
/// This is the microkernel-minimal set — memory + scheduling only; file/device
|
||||||
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
|
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
|
||||||
/// written back into the trap frame, since the entry paths restore user registers
|
/// written back into the trap frame, since the entry paths restore user registers
|
||||||
/// from it. One handler serves both the syscall/sysret and int-0x80 entry paths.
|
/// from it. One handler serves both the system_call/sysret and int-0x80 entry paths.
|
||||||
///
|
///
|
||||||
/// Install it once at boot (before any user code runs) via `init`.
|
/// Install it once at boot (before any user code runs) via `init`.
|
||||||
pub fn init() void {
|
pub fn init() void {
|
||||||
arch.setSyscallHandler(syscall);
|
architecture.setSystemCallHandler(system_call);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return -1 (as an unsigned bit pattern) in the syscall result register.
|
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||||
fn fail(state: *arch.CpuState) void {
|
fn fail(state: *architecture.CpuState) void {
|
||||||
arch.setSyscallResult(state, @bitCast(@as(i64, -1)));
|
architecture.setSystemCallResult(state, @bitCast(@as(i64, -1)));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn syscall(state: *arch.CpuState) void {
|
fn system_call(state: *architecture.CpuState) void {
|
||||||
switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) {
|
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
|
||||||
.exit => {
|
.exit => {
|
||||||
exit_code = arch.syscallArg(state, 0);
|
exit_code = architecture.systemCallArg(state, 0);
|
||||||
// A scheduled process drops its endpoint references, frees its address
|
// A scheduled process drops its endpoint references, frees its address
|
||||||
// space, and reschedules; a borrowed test thread unwinds back to the
|
// space, and reschedules; a borrowed test thread unwinds back to the
|
||||||
// kernel that entered it.
|
// kernel that entered it.
|
||||||
if (sched.currentIsUserProcess()) {
|
if (scheduler.currentIsUserProcess()) {
|
||||||
ipc.closeHandles(sched.cur());
|
// Unbind before closeHandles: dropping the last reference destroys the
|
||||||
sched.exitUser();
|
// Endpoint, and a still-bound GSI would have an ISR call
|
||||||
} else arch.userExit();
|
// notifyFromIsr on freed memory the next time the device fired.
|
||||||
|
// unbindAll also leaves the line masked, so a dead driver's device
|
||||||
|
// goes quiet rather than storming.
|
||||||
|
releaseIrqs(scheduler.current());
|
||||||
|
ipc.closeHandles(scheduler.current());
|
||||||
|
scheduler.exitUser();
|
||||||
|
} else architecture.userExit();
|
||||||
},
|
},
|
||||||
.yield => {
|
.yield => {
|
||||||
sched.yield();
|
scheduler.yield();
|
||||||
arch.setSyscallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
},
|
},
|
||||||
.sleep => {
|
.sleep => {
|
||||||
sched.sleep(arch.syscallArg(state, 0));
|
scheduler.sleep(architecture.systemCallArg(state, 0));
|
||||||
arch.setSyscallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
},
|
},
|
||||||
.debug_write => sysDebugWrite(state),
|
.debug_write => systemDebugWrite(state),
|
||||||
.mmap => sysMmap(state),
|
.mmap => systemMmap(state),
|
||||||
.munmap => sysMunmap(state),
|
.munmap => systemMunmap(state),
|
||||||
.create_endpoint => sysCreateEndpoint(state),
|
.create_endpoint => systemCreateEndpoint(state),
|
||||||
.ipc_register => sysIpcRegister(state),
|
.ipc_register => systemIpcRegister(state),
|
||||||
.ipc_lookup => sysIpcLookup(state),
|
.ipc_lookup => systemIpcLookup(state),
|
||||||
.ipc_call => sysIpcCall(state),
|
.ipc_call => systemIpcCall(state),
|
||||||
.ipc_reply_wait => sysIpcReplyWait(state),
|
.ipc_reply_wait => systemIpcReplyWait(state),
|
||||||
.dev_enumerate => sysDevEnumerate(state),
|
.device_enumerate => systemDeviceEnumerate(state),
|
||||||
.dev_claim => sysDevClaim(state),
|
.device_claim => systemDeviceClaim(state),
|
||||||
.mmio_map => sysMmioMap(state),
|
.mmio_map => systemMmioMap(state),
|
||||||
// irq_bind/irq_ack (IRQ-as-message) land with the driver that needs them.
|
.irq_bind => systemIrqBind(state),
|
||||||
.irq_bind, .irq_ack => fail(state),
|
.irq_ack => systemIrqAck(state),
|
||||||
|
.device_register => systemDeviceRegister(state),
|
||||||
_ => fail(state),
|
_ => fail(state),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return `-errno` in the syscall result register.
|
/// Return `-errno` in the system_call result register.
|
||||||
fn failErr(state: *arch.CpuState, errno: i64) void {
|
fn failErr(state: *architecture.CpuState, errno: i64) void {
|
||||||
arch.setSyscallResult(state, @bitCast(-errno));
|
architecture.setSystemCallResult(state, @bitCast(-errno));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// create_endpoint() -> handle: allocate an endpoint and install it in the
|
/// create_endpoint() -> handle: allocate an endpoint and install it in the
|
||||||
/// caller's handle table.
|
/// caller's handle table.
|
||||||
fn sysCreateEndpoint(state: *arch.CpuState) void {
|
fn systemCreateEndpoint(state: *architecture.CpuState) void {
|
||||||
const ep = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
const endpoint = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
||||||
const h = ipc.installHandle(sched.cur(), ep);
|
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||||
if (h < 0) {
|
if (h < 0) {
|
||||||
ipc.dropRef(ep);
|
ipc.dropRef(endpoint);
|
||||||
return failErr(state, ipc.ENOSPC);
|
return failErr(state, ipc.ENOSPC);
|
||||||
}
|
}
|
||||||
arch.setSyscallResult(state, @intCast(h));
|
architecture.setSystemCallResult(state, @intCast(h));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||||
/// well-known id so other processes can find it.
|
/// well-known id so other processes can find it.
|
||||||
fn sysIpcRegister(state: *arch.CpuState) void {
|
fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||||
const id: u32 = @truncate(arch.syscallArg(state, 0));
|
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||||
const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||||
arch.setSyscallResult(state, @bitCast(ipc.register(id, ep)));
|
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||||
/// handle to it in the caller.
|
/// handle to it in the caller.
|
||||||
fn sysIpcLookup(state: *arch.CpuState) void {
|
fn systemIpcLookup(state: *architecture.CpuState) void {
|
||||||
const id: u32 = @truncate(arch.syscallArg(state, 0));
|
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||||
const ep = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||||
const h = ipc.installHandle(sched.cur(), ep);
|
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||||
if (h < 0) {
|
if (h < 0) {
|
||||||
ipc.dropRef(ep);
|
ipc.dropRef(endpoint);
|
||||||
return failErr(state, ipc.ENOSPC);
|
return failErr(state, ipc.ENOSPC);
|
||||||
}
|
}
|
||||||
arch.setSyscallResult(state, @intCast(h));
|
architecture.setSystemCallResult(state, @intCast(h));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// ipc_call(handle, msg_ptr, msg_len, reply_ptr, reply_cap) -> reply_len.
|
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
|
||||||
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
||||||
/// stack, so it survives the block and receives the result on resume.
|
/// stack, so it survives the block and receives the result on resume.
|
||||||
fn sysIpcCall(state: *arch.CpuState) void {
|
fn systemIpcCall(state: *architecture.CpuState) void {
|
||||||
const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
const r = ipc.call(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4));
|
const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4));
|
||||||
arch.setSyscallResult(state, @bitCast(r));
|
architecture.setSystemCallResult(state, @bitCast(r));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// ipc_reply_wait(handle, reply_ptr, reply_len, recv_ptr, recv_cap) -> recv_len,
|
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
|
||||||
/// with the sender's badge in the secondary result register (rdx).
|
/// with the sender's badge in the secondary result register (rdx).
|
||||||
fn sysIpcReplyWait(state: *arch.CpuState) void {
|
fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
||||||
const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
var badge: u64 = 0;
|
var badge: u64 = 0;
|
||||||
const r = ipc.replyWait(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4), &badge);
|
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge);
|
||||||
arch.setSyscallResult(state, @bitCast(r));
|
architecture.setSystemCallResult(state, @bitCast(r));
|
||||||
arch.setSyscallResult2(state, badge);
|
architecture.setSystemCallResult2(state, badge);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// dev_enumerate(buf, max) -> total: snapshot the device table into the caller's
|
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
||||||
/// buffer (up to `max` entries), returning the total device count.
|
/// buffer (up to `maximum` entries), returning the total device count.
|
||||||
fn sysDevEnumerate(state: *arch.CpuState) void {
|
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||||
const buf_ptr = arch.syscallArg(state, 0);
|
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||||
const max = arch.syscallArg(state, 1);
|
const maximum = architecture.systemCallArg(state, 1);
|
||||||
const t = sched.cur();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or buf_ptr >= user_half_end) return fail(state);
|
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(danos.DeviceDesc);
|
const sz = @sizeOf(danos.DeviceDescriptor);
|
||||||
const cap = @min(max, (user_half_end - buf_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]danos.DeviceDesc = @ptrFromInt(buf_ptr);
|
const out: [*]danos.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||||
arch.setSyscallResult(state, devsvc.enumerate(out[0..@intCast(cap)]));
|
architecture.setSystemCallResult(state, device_service.enumerate(out[0..@intCast(cap)]));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// dev_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||||
fn sysDevClaim(state: *arch.CpuState) void {
|
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||||
if (devsvc.claim(arch.syscallArg(state, 0), sched.cur().id))
|
if (device_service.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||||
arch.setSyscallResult(state, 0)
|
architecture.setSystemCallResult(state, 0)
|
||||||
else
|
else
|
||||||
fail(state);
|
fail(state);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// mmio_map(dev_id, res_idx) -> vaddr: map a claimed device's MMIO window into
|
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||||
/// this address space (strong-uncacheable) and return the register base address.
|
/// this address space (strong-uncacheable) and return the register base address.
|
||||||
/// The claim is the capability — a process can only map hardware it owns.
|
/// The claim is the capability — a process can only map hardware it owns.
|
||||||
fn sysMmioMap(state: *arch.CpuState) void {
|
fn systemMmioMap(state: *architecture.CpuState) void {
|
||||||
const dev_id = arch.syscallArg(state, 0);
|
const device_id = architecture.systemCallArg(state, 0);
|
||||||
const res_idx = arch.syscallArg(state, 1);
|
const resource_index = architecture.systemCallArg(state, 1);
|
||||||
const t = sched.cur();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.aspace == 0) return fail(state);
|
||||||
const owner = devsvc.ownerOf(dev_id) orelse return fail(state);
|
const owner = device_service.ownerOf(device_id) orelse return fail(state);
|
||||||
if (owner != t.id) return fail(state); // not claimed by this process
|
if (owner != t.id) return fail(state); // not claimed by this process
|
||||||
const r = devsvc.resourceOf(dev_id, res_idx) orelse return fail(state);
|
const r = device_service.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||||
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
|
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
|
||||||
|
|
||||||
if (t.dev_map_next == 0) t.dev_map_next = dev_arena_base;
|
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||||
const first = r.start & ~@as(u64, page_size - 1);
|
const first = r.start & ~@as(u64, page_size - 1);
|
||||||
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
||||||
const pages = (last - first) / page_size + 1;
|
const pages = (last - first) / page_size + 1;
|
||||||
const base_v = t.dev_map_next;
|
const base_v = t.device_map_next;
|
||||||
if (base_v + pages * page_size > dev_arena_end) return fail(state);
|
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||||
|
|
||||||
arch.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
|
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
|
||||||
t.dev_map_next = base_v + pages * page_size;
|
t.device_map_next = base_v + pages * page_size;
|
||||||
arch.setSyscallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||||
|
}
|
||||||
|
|
||||||
|
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||||
|
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
||||||
|
/// enumerates it and hands each device it finds to the table, where a class driver
|
||||||
|
/// can claim it.
|
||||||
|
///
|
||||||
|
/// The kernel copies the descriptor into a kernel local *once* (via the same
|
||||||
|
/// physmap-walking path as IPC, so an unmapped user page fails the call rather than
|
||||||
|
/// faulting the kernel), then validates and uses that copy — no second read of user
|
||||||
|
/// memory, so nothing it checked can change under it. It refuses any child resource
|
||||||
|
/// that escapes the parent's windows: a descriptor is a licence to map physical
|
||||||
|
/// memory, so a bus may only subdivide what it already holds. `id`/`parent` in the
|
||||||
|
/// supplied descriptor are ignored.
|
||||||
|
fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||||
|
const parent_id = architecture.systemCallArg(state, 0);
|
||||||
|
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
|
||||||
|
var descriptor: danos.DeviceDescriptor = undefined;
|
||||||
|
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||||
|
|
||||||
|
const id = device_service.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||||
|
architecture.setSystemCallResult(state, id);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
|
||||||
|
/// (which is what frees the endpoints an ISR would otherwise notify into).
|
||||||
|
fn releaseIrqs(t: *scheduler.Task) void {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
irq.releaseOwner(t.id);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||||
|
/// The two checks are the whole security story: the device must be *claimed* by the
|
||||||
|
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
||||||
|
/// by discovery. Neither a raw GSI nor an unclaimed device can get through — which
|
||||||
|
/// is why irq_bind takes a resource index and not an interrupt number.
|
||||||
|
fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||||
|
const owner = device_service.ownerOf(device_id) orelse return null;
|
||||||
|
if (owner != t.id) return null;
|
||||||
|
const r = device_service.resourceOf(device_id, resource_index) orelse return null;
|
||||||
|
if (r.kind != @intFromEnum(danos.ResourceKind.irq)) return null;
|
||||||
|
if (r.start >= irq.maximum_gsi) return null;
|
||||||
|
return @intCast(r.start);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
|
||||||
|
/// endpoint as an asynchronous IPC notification. The driver then blocks in
|
||||||
|
/// IPC_ReplyWait and is woken by the ISR; see src/kernel/irq.zig for the cycle.
|
||||||
|
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||||
|
return fail(state);
|
||||||
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||||
|
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
irq.bind(gsi, endpoint, t.id) catch return fail(state);
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// irq_ack(device_id, resource_index) -> 0/-1: re-arm a bound IRQ. The ISR left the line
|
||||||
|
/// masked (it could not quiet the device — that's this driver's job), so nothing
|
||||||
|
/// more arrives until the driver says it has serviced the hardware.
|
||||||
|
fn systemIrqAck(state: *architecture.CpuState) void {
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||||
|
return fail(state);
|
||||||
|
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
if (irq.ack(gsi)) architecture.setSystemCallResult(state, 0) else fail(state);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
||||||
@@ -243,18 +328,18 @@ fn sysMmioMap(state: *arch.CpuState) void {
|
|||||||
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
||||||
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
||||||
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
||||||
fn sysDebugWrite(state: *arch.CpuState) void {
|
fn systemDebugWrite(state: *architecture.CpuState) void {
|
||||||
const ptr = arch.syscallArg(state, 0);
|
const ptr = architecture.systemCallArg(state, 0);
|
||||||
const len = arch.syscallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
if (len <= write_buf.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||||
const src: [*]const u8 = @ptrFromInt(ptr);
|
const source: [*]const u8 = @ptrFromInt(ptr);
|
||||||
@memcpy(write_buf[0..len], src[0..len]); // keep the latest message
|
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
|
||||||
write_len = len;
|
write_len = len;
|
||||||
write_from_user = arch.fromUser(state);
|
write_from_user = architecture.fromUser(state);
|
||||||
write_count += 1;
|
write_count += 1;
|
||||||
log.write("DANOS-INIT: ");
|
log.write("DANOS-INIT: ");
|
||||||
log.write(src[0..len]);
|
log.write(source[0..len]);
|
||||||
arch.setSyscallResult(state, len);
|
architecture.setSystemCallResult(state, len);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
}
|
}
|
||||||
@@ -264,13 +349,13 @@ fn sysDebugWrite(state: *arch.CpuState) void {
|
|||||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||||
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
|
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
|
||||||
/// -1. The user-space allocator (lib `rt`) carves these pages into malloc blocks.
|
/// -1. The user-space allocator (lib `runtime`) carves these pages into malloc blocks.
|
||||||
fn sysMmap(state: *arch.CpuState) void {
|
fn systemMmap(state: *architecture.CpuState) void {
|
||||||
const len = arch.syscallArg(state, 0);
|
const len = architecture.systemCallArg(state, 0);
|
||||||
const t = sched.cur();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
||||||
const pages = (len + page_size - 1) / page_size;
|
const pages = (len + page_size - 1) / page_size;
|
||||||
if (pages == 0 or pages > max_mmap_pages) return fail(state);
|
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||||
|
|
||||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
||||||
const base = t.heap_next;
|
const base = t.heap_next;
|
||||||
@@ -278,7 +363,7 @@ fn sysMmap(state: *arch.CpuState) void {
|
|||||||
|
|
||||||
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
||||||
// (no partially-mapped grant leaks into the address space).
|
// (no partially-mapped grant leaks into the address space).
|
||||||
var frames: [max_mmap_pages]u64 = undefined;
|
var frames: [maximum_mmap_pages]u64 = undefined;
|
||||||
var got: usize = 0;
|
var got: usize = 0;
|
||||||
while (got < pages) : (got += 1) {
|
while (got < pages) : (got += 1) {
|
||||||
frames[got] = pmm.alloc() orelse {
|
frames[got] = pmm.alloc() orelse {
|
||||||
@@ -288,12 +373,12 @@ fn sysMmap(state: *arch.CpuState) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
for (frames[0..pages], 0..) |frame, i| {
|
for (frames[0..pages], 0..) |frame, i| {
|
||||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||||
@memset(dst[0..page_size], 0); // hand out zeroed memory
|
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||||
arch.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
||||||
}
|
}
|
||||||
t.heap_next = base + pages * page_size;
|
t.heap_next = base + pages * page_size;
|
||||||
arch.setSyscallResult(state, base);
|
architecture.setSystemCallResult(state, base);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||||
@@ -301,25 +386,25 @@ fn sysMmap(state: *arch.CpuState) void {
|
|||||||
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
|
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
|
||||||
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
|
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
|
||||||
/// range is not page-aligned or lies outside the arena.
|
/// range is not page-aligned or lies outside the arena.
|
||||||
fn sysMunmap(state: *arch.CpuState) void {
|
fn systemMunmap(state: *architecture.CpuState) void {
|
||||||
const base = arch.syscallArg(state, 0);
|
const base = architecture.systemCallArg(state, 0);
|
||||||
const len = arch.syscallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
const t = sched.cur();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
||||||
const pages = (len + page_size - 1) / page_size;
|
const pages = (len + page_size - 1) / page_size;
|
||||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||||
|
|
||||||
for (0..pages) |i| {
|
for (0..pages) |i| {
|
||||||
const va = base + i * page_size;
|
const va = base + i * page_size;
|
||||||
if (arch.translate(t.aspace, va)) |phys| {
|
if (architecture.translate(t.aspace, va)) |physical| {
|
||||||
arch.unmapUserPageInto(t.aspace, va);
|
architecture.unmapUserPageInto(t.aspace, va);
|
||||||
pmm.free(phys);
|
pmm.free(physical);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
arch.setSyscallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Reset the recorded syscall evidence before a user-mode run.
|
/// Reset the recorded system_call evidence before a user-mode run.
|
||||||
fn resetRecords() void {
|
fn resetRecords() void {
|
||||||
write_len = 0;
|
write_len = 0;
|
||||||
write_from_user = false;
|
write_from_user = false;
|
||||||
@@ -329,8 +414,8 @@ fn resetRecords() void {
|
|||||||
|
|
||||||
pub const RunError = error{ ProgramTooBig, OutOfMemory };
|
pub const RunError = error{ ProgramTooBig, OutOfMemory };
|
||||||
|
|
||||||
/// Map `blob` at code_virt with a fresh user stack, drop to ring 3, and return
|
/// Map `blob` at code_virtual with a fresh user stack, drop to ring 3, and return
|
||||||
/// once the program exits via syscall 0. See the migration caveat in the module
|
/// once the program exits via system_call 0. See the migration caveat in the module
|
||||||
/// doc. A program that faults instead never returns (on_fault halts the core).
|
/// doc. A program that faults instead never returns (on_fault halts the core).
|
||||||
pub fn run(blob: []const u8) RunError!void {
|
pub fn run(blob: []const u8) RunError!void {
|
||||||
if (blob.len > page_size) return error.ProgramTooBig;
|
if (blob.len > page_size) return error.ProgramTooBig;
|
||||||
@@ -343,20 +428,20 @@ pub fn run(blob: []const u8) RunError!void {
|
|||||||
// Fill the code frame through the physmap (supervisor RW): the user-facing
|
// Fill the code frame through the physmap (supervisor RW): the user-facing
|
||||||
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
|
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
|
||||||
// padded with int3 so a stray jump traps instead of sliding.
|
// padded with int3 so a stray jump traps instead of sliding.
|
||||||
const code: [*]u8 = @ptrFromInt(danos.physToVirt(code_frame));
|
const code: [*]u8 = @ptrFromInt(danos.physicalToVirtual(code_frame));
|
||||||
@memcpy(code[0..blob.len], blob);
|
@memcpy(code[0..blob.len], blob);
|
||||||
@memset(code[blob.len..page_size], 0xCC);
|
@memset(code[blob.len..page_size], 0xCC);
|
||||||
|
|
||||||
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
|
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
|
||||||
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
|
architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX
|
||||||
resetRecords();
|
resetRecords();
|
||||||
|
|
||||||
arch.enterUser(sched.currentCpuIndex(), code_virt, stack_virt + page_size);
|
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size);
|
||||||
|
|
||||||
// Back via the exit syscall; the interrupt gate left IF clear.
|
// Back via the exit system_call; the interrupt gate left IF clear.
|
||||||
arch.enableInterrupts();
|
architecture.enableInterrupts();
|
||||||
arch.unmapPage(code_virt);
|
architecture.unmapPage(code_virtual);
|
||||||
arch.unmapPage(stack_virt);
|
architecture.unmapPage(stack_virtual);
|
||||||
pmm.free(code_frame);
|
pmm.free(code_frame);
|
||||||
pmm.free(stack_frame);
|
pmm.free(stack_frame);
|
||||||
}
|
}
|
||||||
@@ -371,8 +456,8 @@ pub const InitError = error{
|
|||||||
OutOfMemory,
|
OutOfMemory,
|
||||||
};
|
};
|
||||||
|
|
||||||
const max_segments = 16;
|
const maximum_segments = 16;
|
||||||
const max_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||||
|
|
||||||
const Segment = struct {
|
const Segment = struct {
|
||||||
vaddr: u64,
|
vaddr: u64,
|
||||||
@@ -391,7 +476,7 @@ const Segment = struct {
|
|||||||
/// against the image and the user region; segments must be page-aligned,
|
/// against the image and the user region; segments must be page-aligned,
|
||||||
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
|
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
|
||||||
/// segment, mapped RO+NX).
|
/// segment, mapped RO+NX).
|
||||||
fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!struct { count: usize, entry: u64 } {
|
fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!struct { count: usize, entry: u64 } {
|
||||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
|
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
|
||||||
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
|
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
|
||||||
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
|
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
|
||||||
@@ -399,7 +484,7 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru
|
|||||||
if (ehdr.e_machine != .X86_64) return error.BadElf;
|
if (ehdr.e_machine != .X86_64) return error.BadElf;
|
||||||
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
|
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
|
||||||
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
|
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
|
||||||
if (ehdr.e_phnum > max_segments) return error.BadElf;
|
if (ehdr.e_phnum > maximum_segments) return error.BadElf;
|
||||||
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
|
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
|
||||||
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
|
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
|
||||||
|
|
||||||
@@ -415,8 +500,8 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru
|
|||||||
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
|
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
|
||||||
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
|
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
|
||||||
// Inside the user image region, strictly below the stack page.
|
// Inside the user image region, strictly below the stack page.
|
||||||
if (phdr.p_vaddr < code_virt) return error.BadSegment;
|
if (phdr.p_vaddr < code_virtual) return error.BadSegment;
|
||||||
if (phdr.p_memsz > stack_virt - phdr.p_vaddr) return error.BadSegment;
|
if (phdr.p_memsz > stack_virtual - phdr.p_vaddr) return error.BadSegment;
|
||||||
|
|
||||||
const w = phdr.p_flags & elf.PF_W != 0;
|
const w = phdr.p_flags & elf.PF_W != 0;
|
||||||
const x = phdr.p_flags & elf.PF_X != 0;
|
const x = phdr.p_flags & elf.PF_X != 0;
|
||||||
@@ -437,7 +522,7 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru
|
|||||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
||||||
}
|
}
|
||||||
total_pages += seg.pages();
|
total_pages += seg.pages();
|
||||||
if (total_pages > max_pages) return error.ProgramTooBig;
|
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
||||||
segs[count] = seg;
|
segs[count] = seg;
|
||||||
count += 1;
|
count += 1;
|
||||||
}
|
}
|
||||||
@@ -457,38 +542,38 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru
|
|||||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||||
@memset(dst[0..page_size], 0);
|
@memset(destination[0..page_size], 0);
|
||||||
const page_off = page_index * page_size;
|
const page_off = page_index * page_size;
|
||||||
if (page_off < seg.filesz) {
|
if (page_off < seg.filesz) {
|
||||||
const n = @min(page_size, seg.filesz - page_off);
|
const n = @min(page_size, seg.filesz - page_off);
|
||||||
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
|
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
||||||
}
|
}
|
||||||
arch.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
||||||
/// ring-3 process at `priority`. Returns immediately — the process runs
|
/// ring-3 process at `priority`. Returns immediately — the process runs
|
||||||
/// preemptively on its own page tables alongside everything else, and its exit
|
/// preemptively on its own page tables alongside everything else, and its exit
|
||||||
/// is handled by the syscall layer. The whole build (address space + ELF load +
|
/// is handled by the system_call layer. The whole build (address space + ELF load +
|
||||||
/// task) runs under the kernel lock so it appears atomically and can't race
|
/// task) runs under the kernel lock so it appears atomically and can't race
|
||||||
/// pmm/heap on another core.
|
/// pmm/heap on another core.
|
||||||
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
||||||
var segs: [max_segments]Segment = undefined;
|
var segs: [maximum_segments]Segment = undefined;
|
||||||
const parsed = try parseSegments(image, &segs);
|
const parsed = try parseSegments(image, &segs);
|
||||||
|
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
|
||||||
const aspace = arch.createAddressSpace() orelse return error.OutOfMemory;
|
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||||
errdefer arch.destroyAddressSpace(aspace);
|
errdefer architecture.destroyAddressSpace(aspace);
|
||||||
|
|
||||||
for (segs[0..parsed.count]) |seg| {
|
for (segs[0..parsed.count]) |seg| {
|
||||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||||
}
|
}
|
||||||
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
arch.mapUserPageInto(aspace, stack_virt, stack_frame, true, false); // RW + NX
|
architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX
|
||||||
|
|
||||||
if (!sched.spawnUserLocked(aspace, parsed.entry, stack_virt + page_size, priority))
|
if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority))
|
||||||
return error.OutOfMemory;
|
return error.OutOfMemory;
|
||||||
}
|
}
|
||||||
|
|||||||
+78
-78
@@ -18,17 +18,17 @@
|
|||||||
//! shared queues.
|
//! shared queues.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const config = @import("config");
|
const parameters = @import("parameters");
|
||||||
const arch = @import("arch");
|
const architecture = @import("architecture");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
|
|
||||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
||||||
pub const Priority = u3;
|
pub const Priority = u3;
|
||||||
const num_priorities = 8;
|
const number_priorities = 8;
|
||||||
|
|
||||||
const stack_size = config.kernel_stack_size; // each task's kernel stack
|
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
||||||
const max_tasks = config.max_tasks; // maximum tasks alive at once (static pool)
|
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
||||||
|
|
||||||
const State = enum { free, ready, running, blocked };
|
const State = enum { free, ready, running, blocked };
|
||||||
|
|
||||||
@@ -52,11 +52,11 @@ pub const Task = struct {
|
|||||||
heap_next: u64 = 0,
|
heap_next: u64 = 0,
|
||||||
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
||||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||||
dev_map_next: u64 = 0,
|
device_map_next: u64 = 0,
|
||||||
// --- synchronous IPC (ipc_sync.zig) ---
|
// --- synchronous IPC (ipc_sync.zig) ---
|
||||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
||||||
// opaque here so the scheduler and IPC modules don't import each other.
|
// opaque here so the scheduler and IPC modules don't import each other.
|
||||||
handles: [ipc_max_handles]?*anyopaque = .{null} ** ipc_max_handles,
|
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
||||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||||
@@ -71,13 +71,13 @@ pub const Task = struct {
|
|||||||
|
|
||||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||||
pub const ipc_max_handles = 16;
|
pub const ipc_maximum_handles = 16;
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** max_tasks;
|
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||||
var next_id: u32 = 1;
|
var next_id: u32 = 1;
|
||||||
|
|
||||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||||
/// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a
|
/// queue of tasks **pinned** to it. One entry per core; the architecture layer stashes a
|
||||||
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
||||||
/// with a single read and no lock.
|
/// with a single read and no lock.
|
||||||
///
|
///
|
||||||
@@ -95,30 +95,30 @@ pub const PerCpu = struct {
|
|||||||
online: bool = false, // has this core finished bring-up?
|
online: bool = false, // has this core finished bring-up?
|
||||||
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
||||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||||
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
|
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||||
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
|
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||||
pinned_bitmap: u8 = 0,
|
pinned_bitmap: u8 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const max_cpus = config.max_cpus;
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
var cpus = [_]PerCpu{.{}} ** max_cpus;
|
var cpus = [_]PerCpu{.{}} ** maximum_cpus;
|
||||||
|
|
||||||
/// This core's per-CPU state, via the arch layer's GS-base pointer. Valid only once
|
/// This core's per-CPU state, via the architecture layer's GS-base pointer. Valid only once
|
||||||
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
|
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
|
||||||
inline fn thisCpu() *PerCpu {
|
inline fn thisCpu() *PerCpu {
|
||||||
return @ptrFromInt(arch.cpuLocal());
|
return @ptrFromInt(architecture.cpuLocal());
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The task running on this core — the per-CPU replacement for the old global
|
/// The task running on this core — the per-CPU replacement for the old global
|
||||||
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
||||||
pub inline fn cur() *Task {
|
pub inline fn current() *Task {
|
||||||
return thisCpu().current;
|
return thisCpu().current;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
|
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
|
||||||
// are shared across all cores and mutated only under the big kernel lock.
|
// are shared across all cores and mutated only under the big kernel lock.
|
||||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
var ready_head: [number_priorities]?*Task = .{null} ** number_priorities;
|
||||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
var ready_tail: [number_priorities]?*Task = .{null} ** number_priorities;
|
||||||
var ready_bitmap: u8 = 0;
|
var ready_bitmap: u8 = 0;
|
||||||
|
|
||||||
var preemption_enabled = true;
|
var preemption_enabled = true;
|
||||||
@@ -129,12 +129,12 @@ var preemption_enabled = true;
|
|||||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||||
pub fn init(boot_priority: Priority) void {
|
pub fn init(boot_priority: Priority) void {
|
||||||
const pc = &cpus[0];
|
const pc = &cpus[0];
|
||||||
pc.* = .{ .index = 0, .online = true, .loaded_aspace = arch.kernelPageTable() };
|
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
||||||
arch.setCpuLocal(0, @intFromPtr(pc));
|
architecture.setCpuLocal(0, @intFromPtr(pc));
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||||
pc.current = &tasks[0];
|
pc.current = &tasks[0];
|
||||||
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
||||||
arch.setTickHook(tick);
|
architecture.setTickHook(tick);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
||||||
@@ -145,7 +145,7 @@ fn idle() void {
|
|||||||
|
|
||||||
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
||||||
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
|
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
|
||||||
/// pointer the arch bring-up hands to the core (it publishes it in its GS base).
|
/// pointer the architecture bring-up hands to the core (it publishes it in its GS base).
|
||||||
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
||||||
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
||||||
const pc = &cpus[index];
|
const pc = &cpus[index];
|
||||||
@@ -153,12 +153,12 @@ pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
|||||||
return pc;
|
return pc;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Entry for an application processor once the arch layer has set up its per-CPU
|
/// Entry for an application processor once the architecture layer has set up its per-CPU
|
||||||
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
||||||
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
||||||
/// interrupts enabled the timer preempts this idle context into whatever the global
|
/// interrupts enabled the timer preempts this idle context into whatever the global
|
||||||
/// ready queue offers, so the core runs real work in parallel with the others. The
|
/// ready queue offers, so the core runs real work in parallel with the others. The
|
||||||
/// `.c` calling convention lets the arch trampoline path jump here. Never returns.
|
/// `.c` calling convention lets the architecture trampoline path jump here. Never returns.
|
||||||
pub fn secondaryMain() callconv(.c) noreturn {
|
pub fn secondaryMain() callconv(.c) noreturn {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
@@ -168,10 +168,10 @@ pub fn secondaryMain() callconv(.c) noreturn {
|
|||||||
pc.current = t;
|
pc.current = t;
|
||||||
pc.idle = t;
|
pc.idle = t;
|
||||||
pc.online = true;
|
pc.online = true;
|
||||||
pc.loaded_aspace = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
|
|
||||||
arch.enableInterrupts(); // the timer now preempts this idle context into work
|
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
||||||
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -195,7 +195,7 @@ fn enqueue(t: *Task) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
fn enqueueTo(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
||||||
t.next = null;
|
t.next = null;
|
||||||
const p: usize = t.priority;
|
const p: usize = t.priority;
|
||||||
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
||||||
@@ -206,7 +206,7 @@ fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitma
|
|||||||
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
||||||
fn topLevel(bitmap: u8) i32 {
|
fn topLevel(bitmap: u8) i32 {
|
||||||
if (bitmap == 0) return -1;
|
if (bitmap == 0) return -1;
|
||||||
return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap));
|
return @as(i32, number_priorities - 1) - @as(i32, @clz(bitmap));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
||||||
@@ -220,7 +220,7 @@ fn dequeueHighest(pc: *PerCpu) ?*Task {
|
|||||||
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
fn dequeueFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
||||||
const t = head[level].?;
|
const t = head[level].?;
|
||||||
head[level] = t.next;
|
head[level] = t.next;
|
||||||
if (head[level] == null) {
|
if (head[level] == null) {
|
||||||
@@ -249,7 +249,7 @@ pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
|||||||
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
const ok = cpu < max_cpus and cpus[cpu].online;
|
const ok = cpu < maximum_cpus and cpus[cpu].online;
|
||||||
_ = create(entry, priority, if (ok) cpu else null);
|
_ = create(entry, priority, if (ok) cpu else null);
|
||||||
return ok;
|
return ok;
|
||||||
}
|
}
|
||||||
@@ -277,7 +277,7 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
|||||||
t.kstack_top = top;
|
t.kstack_top = top;
|
||||||
// First switch-in lands in startUserTask (no register smuggling — it reads
|
// First switch-in lands in startUserTask (no register smuggling — it reads
|
||||||
// the user entry/stack from the Task itself).
|
// the user entry/stack from the Task itself).
|
||||||
t.sp = arch.initTaskStack(top, @intFromPtr(&startUserTask));
|
t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||||
enqueue(t);
|
enqueue(t);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -287,10 +287,10 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
|||||||
/// Task avoids smuggling values through callee-saved registers across the
|
/// Task avoids smuggling values through callee-saved registers across the
|
||||||
/// context switch and lock release.
|
/// context switch and lock release.
|
||||||
fn startUserTask() void {
|
fn startUserTask() void {
|
||||||
const t = cur();
|
const t = current();
|
||||||
var buf: [96]u8 = undefined;
|
var buffer: [96]u8 = undefined;
|
||||||
arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
||||||
arch.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||||
@@ -303,7 +303,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
|||||||
next_id += 1;
|
next_id += 1;
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
t.kstack_top = top;
|
t.kstack_top = top;
|
||||||
t.sp = arch.initTaskStack(top, @intFromPtr(entry));
|
t.sp = architecture.initTaskStack(top, @intFromPtr(entry));
|
||||||
enqueue(t);
|
enqueue(t);
|
||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
@@ -318,22 +318,22 @@ fn freeSlot() ?*Task {
|
|||||||
/// Pick the highest-priority ready task and switch this core to it. The big kernel
|
/// Pick the highest-priority ready task and switch this core to it. The big kernel
|
||||||
/// lock must be held by the caller (which also keeps local interrupts disabled);
|
/// lock must be held by the caller (which also keeps local interrupts disabled);
|
||||||
/// it serialises every core's scheduling, so no other core can touch the shared
|
/// it serialises every core's scheduling, so no other core can touch the shared
|
||||||
/// queues while we requeue `prev` and dequeue `next`. A dequeued task is `.ready`,
|
/// queues while we requeue `previous` and dequeue `next`. A dequeued task is `.ready`,
|
||||||
/// never running elsewhere, so two cores never run the same task.
|
/// never running elsewhere, so two cores never run the same task.
|
||||||
fn schedule() void {
|
fn schedule() void {
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
const prev = pc.current;
|
const previous = pc.current;
|
||||||
if (prev.state == .running) {
|
if (previous.state == .running) {
|
||||||
prev.state = .ready;
|
previous.state = .ready;
|
||||||
enqueue(prev); // back of its level's queue (round-robin)
|
enqueue(previous); // back of its level's queue (round-robin)
|
||||||
}
|
}
|
||||||
const next = dequeueHighest(pc) orelse {
|
const next = dequeueHighest(pc) orelse {
|
||||||
prev.state = .running; // nothing else ready — keep running
|
previous.state = .running; // nothing else ready — keep running
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
next.state = .running;
|
next.state = .running;
|
||||||
pc.current = next;
|
pc.current = next;
|
||||||
if (next != prev) switchTo(pc, &prev.sp, next);
|
if (next != previous) switchTo(pc, &previous.sp, next);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||||
@@ -346,13 +346,13 @@ fn schedule() void {
|
|||||||
/// interrupt can observe a half-updated (kernel stack, address space) pair.
|
/// interrupt can observe a half-updated (kernel stack, address space) pair.
|
||||||
/// `save_sp` receives the outgoing task's stack pointer.
|
/// `save_sp` receives the outgoing task's stack pointer.
|
||||||
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||||
if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top);
|
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
||||||
const want = if (next.aspace != 0) next.aspace else arch.kernelPageTable();
|
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
||||||
if (want != pc.loaded_aspace) {
|
if (want != pc.loaded_aspace) {
|
||||||
arch.loadPageTable(want);
|
architecture.loadPageTable(want);
|
||||||
pc.loaded_aspace = want;
|
pc.loaded_aspace = want;
|
||||||
}
|
}
|
||||||
arch.switchContext(save_sp, next.sp);
|
architecture.switchContext(save_sp, next.sp);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Voluntarily give up the CPU to the next ready task.
|
/// Voluntarily give up the CPU to the next ready task.
|
||||||
@@ -366,8 +366,8 @@ pub fn yield() void {
|
|||||||
/// again. The idle task (or other work) runs in the meantime.
|
/// again. The idle task (or other work) runs in the meantime.
|
||||||
pub fn sleep(ms: u64) void {
|
pub fn sleep(ms: u64) void {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
const t = cur();
|
const t = current();
|
||||||
t.wake_at = arch.millis() + ms;
|
t.wake_at = architecture.millis() + ms;
|
||||||
t.state = .blocked;
|
t.state = .blocked;
|
||||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
@@ -384,37 +384,37 @@ pub const WaitQueue = struct {
|
|||||||
head: ?*Task = null,
|
head: ?*Task = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Block the current task on `wq` and switch away. Precondition: the big kernel
|
/// Block the current task on `wait_queue` and switch away. Precondition: the big kernel
|
||||||
/// lock is held (so a condition can be checked and the block committed atomically;
|
/// lock is held (so a condition can be checked and the block committed atomically;
|
||||||
/// it also keeps local interrupts disabled). On return — when woken — the lock is
|
/// it also keeps local interrupts disabled). On return — when woken — the lock is
|
||||||
/// still held.
|
/// still held.
|
||||||
pub fn waitLocked(wq: *WaitQueue) void {
|
pub fn waitLocked(wait_queue: *WaitQueue) void {
|
||||||
const t = cur();
|
const t = current();
|
||||||
t.state = .blocked;
|
t.state = .blocked;
|
||||||
t.next = wq.head;
|
t.next = wait_queue.head;
|
||||||
wq.head = t;
|
wait_queue.head = t;
|
||||||
schedule();
|
schedule();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
/// Move the highest-priority waiter on `wait_queue` (if any) to the ready queue.
|
||||||
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
|
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
|
||||||
pub fn wakeLocked(wq: *WaitQueue) void {
|
pub fn wakeLocked(wait_queue: *WaitQueue) void {
|
||||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
// Find the highest-priority waiter (bounded scan) and unlink it.
|
||||||
var best_prev: ?*Task = null;
|
var best_previous: ?*Task = null;
|
||||||
var best: ?*Task = null;
|
var best: ?*Task = null;
|
||||||
var prev: ?*Task = null;
|
var previous: ?*Task = null;
|
||||||
var node = wq.head;
|
var node = wait_queue.head;
|
||||||
while (node) |t| : ({
|
while (node) |t| : ({
|
||||||
prev = t;
|
previous = t;
|
||||||
node = t.next;
|
node = t.next;
|
||||||
}) {
|
}) {
|
||||||
if (best == null or t.priority > best.?.priority) {
|
if (best == null or t.priority > best.?.priority) {
|
||||||
best = t;
|
best = t;
|
||||||
best_prev = prev;
|
best_previous = previous;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
const t = best orelse return;
|
const t = best orelse return;
|
||||||
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
if (best_previous) |p| p.next = t.next else wait_queue.head = t.next;
|
||||||
t.state = .ready;
|
t.state = .ready;
|
||||||
enqueue(t);
|
enqueue(t);
|
||||||
}
|
}
|
||||||
@@ -424,7 +424,7 @@ pub fn wakeLocked(wq: *WaitQueue) void {
|
|||||||
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
||||||
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
|
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
|
||||||
pub fn blockCurrentLocked() void {
|
pub fn blockCurrentLocked() void {
|
||||||
cur().state = .blocked;
|
current().state = .blocked;
|
||||||
schedule();
|
schedule();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -436,18 +436,18 @@ pub fn readyLocked(t: *Task) void {
|
|||||||
enqueue(t);
|
enqueue(t);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Block on `wq` (a self-contained critical section).
|
/// Block on `wait_queue` (a self-contained critical section).
|
||||||
pub fn wait(wq: *WaitQueue) void {
|
pub fn wait(wait_queue: *WaitQueue) void {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
waitLocked(wq);
|
waitLocked(wait_queue);
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
/// Wake the highest-priority waiter on `wait_queue`, preempting if it outranks us.
|
||||||
pub fn wake(wq: *WaitQueue) void {
|
pub fn wake(wait_queue: *WaitQueue) void {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
wakeLocked(wq);
|
wakeLocked(wait_queue);
|
||||||
// If a task this core would now pick outranks the running one, run it at once.
|
// If a task this core would now pick outranks the running one, run it at once.
|
||||||
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
||||||
// next tick; this core doesn't preempt for work it can't run.)
|
// next tick; this core doesn't preempt for work it can't run.)
|
||||||
@@ -468,7 +468,7 @@ fn highestReadyPriority(pc: *PerCpu) ?Priority {
|
|||||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
||||||
fn wakeExpired() void {
|
fn wakeExpired() void {
|
||||||
const now = arch.millis();
|
const now = architecture.millis();
|
||||||
for (&tasks) |*t| {
|
for (&tasks) |*t| {
|
||||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
||||||
t.wake_at = 0;
|
t.wake_at = 0;
|
||||||
@@ -521,10 +521,10 @@ pub fn exitUser() noreturn {
|
|||||||
const dying = pc.current;
|
const dying = pc.current;
|
||||||
const as = dying.aspace;
|
const as = dying.aspace;
|
||||||
if (as != 0) {
|
if (as != 0) {
|
||||||
const kroot = arch.kernelPageTable();
|
const kroot = architecture.kernelPageTable();
|
||||||
arch.loadPageTable(kroot); // off the process tables before freeing them
|
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||||
pc.loaded_aspace = kroot;
|
pc.loaded_aspace = kroot;
|
||||||
arch.destroyAddressSpace(as);
|
architecture.destroyAddressSpace(as);
|
||||||
}
|
}
|
||||||
dying.state = .free;
|
dying.state = .free;
|
||||||
dying.aspace = 0;
|
dying.aspace = 0;
|
||||||
@@ -538,11 +538,11 @@ pub fn exitUser() noreturn {
|
|||||||
|
|
||||||
/// Whether the running task is a user process (has its own address space).
|
/// Whether the running task is a user process (has its own address space).
|
||||||
pub fn currentIsUserProcess() bool {
|
pub fn currentIsUserProcess() bool {
|
||||||
return cur().aspace != 0;
|
return current().aspace != 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
pub fn currentId() u32 {
|
||||||
return cur().id;
|
return current().id;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
||||||
@@ -551,11 +551,11 @@ pub fn currentId() u32 {
|
|||||||
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
||||||
/// reporter can call it unconditionally without a second fault.
|
/// reporter can call it unconditionally without a second fault.
|
||||||
pub fn currentCpuIndex() u32 {
|
pub fn currentCpuIndex() u32 {
|
||||||
if (arch.cpuLocal() == 0) return 0;
|
if (architecture.cpuLocal() == 0) return 0;
|
||||||
return thisCpu().index;
|
return thisCpu().index;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||||
pub fn setPriority(p: Priority) void {
|
pub fn setPriority(p: Priority) void {
|
||||||
cur().priority = p;
|
current().priority = p;
|
||||||
}
|
}
|
||||||
|
|||||||
+4
-4
@@ -28,7 +28,7 @@
|
|||||||
//! `releaseForFreshTask` before running the task body.
|
//! `releaseForFreshTask` before running the task body.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const arch = @import("arch");
|
const architecture = @import("architecture");
|
||||||
|
|
||||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||||
var held = std.atomic.Value(u32).init(0);
|
var held = std.atomic.Value(u32).init(0);
|
||||||
@@ -38,7 +38,7 @@ var held = std.atomic.Value(u32).init(0);
|
|||||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||||
/// can't try to re-acquire the lock we're holding.
|
/// can't try to re-acquire the lock we're holding.
|
||||||
pub fn enter() u64 {
|
pub fn enter() u64 {
|
||||||
const flags = arch.saveInterrupts();
|
const flags = architecture.saveInterrupts();
|
||||||
acquire();
|
acquire();
|
||||||
return flags;
|
return flags;
|
||||||
}
|
}
|
||||||
@@ -48,7 +48,7 @@ pub fn enter() u64 {
|
|||||||
/// section reached from task context (`yield`, `sleep`, `wait`, `wake`, IPC).
|
/// section reached from task context (`yield`, `sleep`, `wait`, `wake`, IPC).
|
||||||
pub fn leave(flags: u64) void {
|
pub fn leave(flags: u64) void {
|
||||||
release();
|
release();
|
||||||
arch.restoreInterrupts(flags);
|
architecture.restoreInterrupts(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Release the lock but leave interrupts as they are. The exit for a critical
|
/// Release the lock but leave interrupts as they are. The exit for a critical
|
||||||
@@ -71,7 +71,7 @@ fn acquire() void {
|
|||||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||||
while (held.swap(1, .acquire) != 0) {
|
while (held.swap(1, .acquire) != 0) {
|
||||||
while (held.load(.monotonic) != 0) arch.cpuRelax();
|
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+428
-221
File diff suppressed because it is too large
Load Diff
@@ -13,11 +13,11 @@
|
|||||||
/// stacks) are allocated at bring-up for cores that actually come online, so this
|
/// stacks) are allocated at bring-up for cores that actually come online, so this
|
||||||
/// ceiling is cheap. A machine with more logical CPUs has its surplus reported and
|
/// ceiling is cheap. A machine with more logical CPUs has its surplus reported and
|
||||||
/// left parked (see acpi `cpusDropped`).
|
/// left parked (see acpi `cpusDropped`).
|
||||||
pub const max_cpus = 128;
|
pub const maximum_cpus = 128;
|
||||||
|
|
||||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP.
|
/// online core consumes one slot for its idle task, plus task 0 on the BSP.
|
||||||
pub const max_tasks = 16;
|
pub const maximum_tasks = 16;
|
||||||
|
|
||||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||||
pub const kernel_stack_size = 16 * 1024;
|
pub const kernel_stack_size = 16 * 1024;
|
||||||
+54
-35
@@ -6,10 +6,10 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
|
||||||
/// Calling convention for the bootloader→kernel jump. Pinned to SysV so it does
|
/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does
|
||||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||||
/// SysV (first arg in RDI). Both reference this to agree on where `*BootInfo`
|
/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation`
|
||||||
/// is passed.
|
/// is passed.
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
|
|
||||||
@@ -47,23 +47,23 @@ pub const page_size = 4096;
|
|||||||
|
|
||||||
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||||
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||||
/// device MMIO windows) is also mapped at `physmap_base + phys`, so the kernel
|
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||||
/// can reach any physical address by adding a constant. The low half is left
|
/// can reach any physical address by adding a constant. The low half is left
|
||||||
/// entirely to user space.
|
/// entirely to user space.
|
||||||
///
|
///
|
||||||
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
||||||
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
||||||
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + phys
|
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical
|
||||||
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
||||||
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||||
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||||
|
|
||||||
/// The kernel syscall numbers — the single source of truth shared by the kernel
|
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||||
/// dispatcher (src/kernel/process.zig) and the user runtime library, so the two
|
/// dispatcher (src/kernel/process.zig) and the user runtime library, so the two
|
||||||
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||||
/// is not here — it lives in user-space servers reached through the IPC calls.
|
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||||
/// The table grows one milestone at a time; see docs/syscall.md.
|
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||||
pub const Syscall = enum(u64) {
|
pub const SystemCall = enum(u64) {
|
||||||
exit = 0, // exit(code): end the calling process
|
exit = 0, // exit(code): end the calling process
|
||||||
yield = 1, // yield(): give up the rest of this quantum
|
yield = 1, // yield(): give up the rest of this quantum
|
||||||
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||||
@@ -73,18 +73,26 @@ pub const Syscall = enum(u64) {
|
|||||||
create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint
|
create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint
|
||||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||||
ipc_call = 9, // ipc_call(h, msg, len, reply, cap) -> reply_len: send + block for reply
|
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, recv, cap) -> recv_len (+badge in rdx)
|
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||||
dev_enumerate = 11, // dev_enumerate(buf, max) -> count: snapshot the device table
|
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||||
dev_claim = 12, // dev_claim(id) -> ok: take exclusive ownership of a device
|
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||||
mmio_map = 13, // mmio_map(id, res_idx) -> vaddr: map a claimed device's MMIO into this AS
|
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||||
irq_bind = 14, // irq_bind(id, res_idx, endpoint): deliver a device IRQ as an IPC notification
|
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||||
irq_ack = 15, // irq_ack(id, res_idx): re-arm a bound IRQ after servicing it
|
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||||
|
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// A device class, mirroring src/device/device.zig's `DeviceClass` **in order**
|
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||||
/// (its `@intFromEnum` values cross the syscall boundary in `DeviceDesc.class`).
|
/// **asynchronous notification** (today: a device interrupt bound with `irq_bind`)
|
||||||
|
/// rather than a message from a client. There is no payload and no reply owed; the
|
||||||
|
/// low bits carry the source, a GSI. Shared so the kernel's ISR and the driver's
|
||||||
|
/// event loop can't disagree about which bit means "the hardware spoke".
|
||||||
|
pub const notify_badge_bit: u64 = 1 << 63;
|
||||||
|
|
||||||
|
/// A device class, mirroring src/device/device-model.zig's `DeviceClass` **in order**
|
||||||
|
/// (its `@intFromEnum` values cross the system_call boundary in `DeviceDescriptor.class`).
|
||||||
/// Keep the two in sync.
|
/// Keep the two in sync.
|
||||||
pub const DeviceClass = enum(u32) {
|
pub const DeviceClass = enum(u32) {
|
||||||
root,
|
root,
|
||||||
@@ -97,7 +105,7 @@ pub const DeviceClass = enum(u32) {
|
|||||||
unknown,
|
unknown,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// A resource kind, mirroring src/device/device.zig's `ResourceKind` in order.
|
/// A resource kind, mirroring src/device/device-model.zig's `ResourceKind` in order.
|
||||||
pub const ResourceKind = enum(u32) {
|
pub const ResourceKind = enum(u32) {
|
||||||
memory,
|
memory,
|
||||||
io_port,
|
io_port,
|
||||||
@@ -106,23 +114,34 @@ pub const ResourceKind = enum(u32) {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// One device resource, as handed to a user-space driver (flat, extern).
|
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||||
pub const ResDesc = extern struct {
|
pub const ResourceDescriptor = extern struct {
|
||||||
kind: u64, // a ResourceKind value
|
kind: u64, // a ResourceKind value
|
||||||
start: u64,
|
start: u64,
|
||||||
len: u64,
|
len: u64,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub const max_dev_resources = 8;
|
pub const maximum_device_resources = 8;
|
||||||
|
|
||||||
/// A device, as snapshotted for user space by `dev_enumerate`. A driver scans
|
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||||
|
pub const no_parent: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||||
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||||
pub const DeviceDesc = extern struct {
|
///
|
||||||
|
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||||
|
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||||
|
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||||
|
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||||
|
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||||
|
/// memory.
|
||||||
|
pub const DeviceDescriptor = extern struct {
|
||||||
id: u64,
|
id: u64,
|
||||||
|
parent: u64, // a device id, or `no_parent`
|
||||||
class: u64, // a DeviceClass value
|
class: u64, // a DeviceClass value
|
||||||
hid_len: u64,
|
hid_len: u64,
|
||||||
resource_count: u64,
|
resource_count: u64,
|
||||||
hid: [8]u8,
|
hid: [8]u8,
|
||||||
resources: [max_dev_resources]ResDesc,
|
resources: [maximum_device_resources]ResourceDescriptor,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
|
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
|
||||||
@@ -145,21 +164,21 @@ pub const prot_exec: u64 = 4;
|
|||||||
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
||||||
/// but must not call this to dereference memory before its own CR3 is loaded —
|
/// but must not call this to dereference memory before its own CR3 is loaded —
|
||||||
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
||||||
pub inline fn physToVirt(phys: u64) u64 {
|
pub inline fn physicalToVirtual(physical: u64) u64 {
|
||||||
return phys + physmap_base;
|
return physical + physmap_base;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Physmap virtual address -> physical. Inverse of `physToVirt`; for producing
|
/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing
|
||||||
/// the physical address of something the kernel holds a physmap pointer to
|
/// the physical address of something the kernel holds a physmap pointer to
|
||||||
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
||||||
pub inline fn virtToPhys(virt: u64) u64 {
|
pub inline fn virtualToPhysical(virtual: u64) u64 {
|
||||||
return virt - physmap_base;
|
return virtual - physmap_base;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// danos's own classification of a span of physical memory — deliberately not
|
/// danos's own classification of a span of physical memory — deliberately not
|
||||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||||
/// native memory description into these kinds, so the kernel never learns what
|
/// native memory description into these kinds, so the kernel never learns what
|
||||||
/// booted it. [[arch]] keeps the same discipline for CPU code.
|
/// booted it. [[architecture]] keeps the same discipline for CPU code.
|
||||||
pub const MemoryKind = enum(u32) {
|
pub const MemoryKind = enum(u32) {
|
||||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||||
@@ -174,7 +193,7 @@ pub const MemoryKind = enum(u32) {
|
|||||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||||
acpi_nvs,
|
acpi_nvs,
|
||||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||||
/// address-space window (e.g. PCIe config space). Kept distinct from
|
/// address-space window (e.g. PCIe configuration space). Kept distinct from
|
||||||
/// `reserved` so RAM accounting doesn't count device address space.
|
/// `reserved` so RAM accounting doesn't count device address space.
|
||||||
mmio,
|
mmio,
|
||||||
};
|
};
|
||||||
@@ -199,20 +218,20 @@ pub const MemoryMap = extern struct {
|
|||||||
|
|
||||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virt` is the higher-half link address;
|
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address;
|
||||||
/// `phys` is where the loader actually placed the segment (they differ once the
|
/// `physical` is where the loader actually placed the segment (they differ once the
|
||||||
/// kernel links high — the loader records the real load address here).
|
/// kernel links high — the loader records the real load address here).
|
||||||
pub const KernelSegment = extern struct {
|
pub const KernelSegment = extern struct {
|
||||||
virt: u64,
|
virtual: u64,
|
||||||
phys: u64,
|
physical: u64,
|
||||||
pages: u64,
|
pages: u64,
|
||||||
flags: u32,
|
flags: u32,
|
||||||
_pad: u32 = 0,
|
_pad: u32 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||||
/// `_start` in RDI (the first argument under the SysV AMD64 C ABI).
|
/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI).
|
||||||
pub const BootInfo = extern struct {
|
pub const BootInformation = extern struct {
|
||||||
framebuffer: Framebuffer,
|
framebuffer: Framebuffer,
|
||||||
memory_map: MemoryMap,
|
memory_map: MemoryMap,
|
||||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||||
@@ -231,7 +250,7 @@ pub const BootInfo = extern struct {
|
|||||||
init_len: u64 = 0,
|
init_len: u64 = 0,
|
||||||
/// The initrd image (a bundle of extra user binaries — the VFS server and
|
/// The initrd image (a bundle of extra user binaries — the VFS server and
|
||||||
/// device drivers), read off the boot volume into memory that survives the
|
/// device drivers), read off the boot volume into memory that survives the
|
||||||
/// handoff, same as `init` above. 0/0 = no initrd. See src/user/proto/initrd.zig.
|
/// handoff, same as `init` above. 0/0 = no initrd. See src/user/protocol/initrd.zig.
|
||||||
initrd_base: u64 = 0,
|
initrd_base: u64 = 0,
|
||||||
initrd_len: u64 = 0,
|
initrd_len: u64 = 0,
|
||||||
};
|
};
|
||||||
|
|||||||
+15
-2
@@ -188,11 +188,24 @@ CASES = [
|
|||||||
{"name": "vfs",
|
{"name": "vfs",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# IO passthrough: a user-space HPET driver maps device MMIO into its own
|
# IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into
|
||||||
# address space and reads the counter advancing.
|
# its own address space, binds the device's interrupt to an IPC endpoint, and
|
||||||
|
# is woken by the hardware five times while blocked (never polling).
|
||||||
{"name": "hpet",
|
{"name": "hpet",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Bus driver: a user process claims a device, enumerates its children from the
|
||||||
|
# hardware, and publishes each with dev_register — and the kernel refuses a child
|
||||||
|
# whose window escapes the parent's (else dev_register maps arbitrary memory).
|
||||||
|
{"name": "bus",
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||||
|
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||||
|
# keeps its own binding. The path hpetd never takes, since it runs forever.
|
||||||
|
{"name": "irqfree",
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The MMIO-grant teardown fix: a device-granted frame must not be reclaimed
|
# The MMIO-grant teardown fix: a device-granted frame must not be reclaimed
|
||||||
# as RAM when its address space is destroyed.
|
# as RAM when its address space is destroyed.
|
||||||
{"name": "iopass",
|
{"name": "iopass",
|
||||||
|
|||||||
Reference in New Issue
Block a user