Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f |
+12
-3
@@ -2,6 +2,7 @@ const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
const EdidActive = uefi.protocol.edid.Active;
|
||||
@@ -84,7 +85,7 @@ fn boot() !noreturn {
|
||||
// the map and exiting would invalidate the map key.
|
||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||
|
||||
log("EFI: kernel loaded, exiting boot services\r\n");
|
||||
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||
boot_information.memory_map = try exitBootServices(bs);
|
||||
|
||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||
@@ -395,7 +396,7 @@ fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
log("EFI: /system/services/init loaded\r\n");
|
||||
progress("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
@@ -403,7 +404,7 @@ fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInfo
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
log("EFI: initial_ramdisk loaded\r\n");
|
||||
progress("EFI: initial_ramdisk loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
@@ -561,6 +562,14 @@ fn log(comptime message: []const u8) void {
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||
}
|
||||
|
||||
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||
/// `log` directly and always show, so a failed boot still explains itself.
|
||||
fn progress(comptime message: []const u8) void {
|
||||
if (!build_options.serial) return;
|
||||
log(message);
|
||||
}
|
||||
|
||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||
fn logBytes(bytes: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
|
||||
@@ -96,6 +96,105 @@ fn addUserBinary(
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||
const KernelModules = struct {
|
||||
boot_handoff: *std.Build.Module,
|
||||
abi: *std.Build.Module,
|
||||
device_abi: *std.Build.Module,
|
||||
architecture: *std.Build.Module,
|
||||
platform: *std.Build.Module,
|
||||
parameters: *std.Build.Module,
|
||||
initial_ramdisk: *std.Build.Module,
|
||||
};
|
||||
|
||||
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||
/// build option baked into `build_options`.
|
||||
fn addKernel(
|
||||
b: *std.Build,
|
||||
kernel_target: std.Build.ResolvedTarget,
|
||||
optimize: std.builtin.OptimizeMode,
|
||||
modules: KernelModules,
|
||||
test_case: ?[]const u8,
|
||||
serial: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
build_options.addOption(bool, "serial", serial);
|
||||
const build_options_module = build_options.createModule();
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||
.{ .name = "abi", .module = modules.abi },
|
||||
.{ .name = "device-abi", .module = modules.device_abi },
|
||||
.{ .name = "architecture", .module = modules.architecture },
|
||||
.{ .name = "platform", .module = modules.platform },
|
||||
.{ .name = "parameters", .module = modules.parameters },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||
/// bundle its own kernel while sharing the (serial-independent) loader, init, and
|
||||
/// ramdisk. Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
init_bin: std.Build.LazyPath,
|
||||
initial_ramdisk_img: std.Build.LazyPath,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_bin);
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
return fat_image;
|
||||
}
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
ensureZigVersion();
|
||||
|
||||
@@ -285,9 +384,12 @@ pub fn build(b: *std.Build) void {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
const build_options_module = build_options.createModule();
|
||||
// The serial-console log sink. Off by default: a real machine often has no
|
||||
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||
|
||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||
@@ -299,41 +401,17 @@ pub fn build(b: *std.Build) void {
|
||||
.abi = .none,
|
||||
});
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "architecture", .module = architecture_module },
|
||||
.{ .name = "platform", .module = platform_module },
|
||||
.{ .name = "parameters", .module = parameters_module },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
const kernel_modules = KernelModules{
|
||||
.boot_handoff = boot_handoff_module,
|
||||
.abi = abi_module,
|
||||
.device_abi = device_abi_module,
|
||||
.architecture = architecture_module,
|
||||
.platform = platform_module,
|
||||
.parameters = parameters_module,
|
||||
.initial_ramdisk = initial_ramdisk_module,
|
||||
};
|
||||
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
@@ -498,6 +576,13 @@ pub fn build(b: *std.Build) void {
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||
// which firmware may mirror to a serial console) are silenced by default — a
|
||||
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||
const loader_options = b.addOptions();
|
||||
loader_options.addOption(bool, "serial", serial);
|
||||
const loader_options_module = loader_options.createModule();
|
||||
const efiexe = b.addExecutable(.{
|
||||
.name = "BOOTX64",
|
||||
.root_module = b.createModule(.{
|
||||
@@ -510,6 +595,7 @@ pub fn build(b: *std.Build) void {
|
||||
.imports = &.{
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "build_options", .module = loader_options_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -525,21 +611,17 @@ pub fn build(b: *std.Build) void {
|
||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efiexe.getEmittedBin());
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(exe.getEmittedBin());
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_exe.getEmittedBin());
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
@@ -611,8 +693,9 @@ pub fn build(b: *std.Build) void {
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image);
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
@@ -634,8 +717,10 @@ pub fn build(b: *std.Build) void {
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||
run_efi.step.dependOn(b.getInstallStep());
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
|
||||
@@ -27,6 +27,15 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||
so a serial-enabled image is still safe on real hardware.)
|
||||
|
||||
## In-kernel test cases
|
||||
|
||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||
|
||||
@@ -38,6 +38,13 @@ pub const Device = struct {
|
||||
return self.transfer(.write, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
|
||||
+28
-62
@@ -3,11 +3,13 @@
|
||||
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||
//! bootloader handed us) and translates the static tables into the generic
|
||||
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
||||
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
||||
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
||||
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
||||
//! interpretation is a separate, larger subproject.
|
||||
//! source. This is deliberately the *static-table* path, and **only** that: MADT
|
||||
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
|
||||
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
|
||||
//! interpreted here — the kernel collects the blobs and publishes them on the
|
||||
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
|
||||
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
|
||||
//! off the single-core critical path (nothing else runs alongside it there).
|
||||
//!
|
||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||
@@ -19,7 +21,6 @@ const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const parameters = @import("parameters");
|
||||
const device_model = @import("device-model.zig");
|
||||
const aml = @import("aml/aml.zig");
|
||||
const DeviceTree = device_model.DeviceTree;
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
@@ -37,8 +38,11 @@ pub const RegisterAccess = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||
/// The power register map, extracted from the FADT during discovery. Populated by
|
||||
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
|
||||
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
|
||||
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
|
||||
/// writes the PM1 control register itself). So this holds only the FADT scalars.
|
||||
pub const PowerInformation = struct {
|
||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||
@@ -54,10 +58,6 @@ pub const PowerInformation = struct {
|
||||
reset: RegisterAccess = .{},
|
||||
reset_value: u8 = 0,
|
||||
reset_supported: bool = false,
|
||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
||||
/// packages.
|
||||
s5: ?aml.SleepType = null,
|
||||
s3: ?aml.SleepType = null,
|
||||
};
|
||||
|
||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||
@@ -141,19 +141,6 @@ const maximum_cpus = parameters.maximum_cpus;
|
||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||
pub var cpu_information: CpuInformation = .{};
|
||||
|
||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
||||
pub const AmlStats = struct {
|
||||
nodes: usize = 0,
|
||||
consumed: usize = 0,
|
||||
total: usize = 0,
|
||||
};
|
||||
pub var aml_stats: AmlStats = .{};
|
||||
|
||||
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
||||
/// device enumeration later. Null until `discover` runs successfully.
|
||||
pub var namespace: ?aml.Namespace = null;
|
||||
|
||||
/// Physical address of the DSDT the FADT points at, or 0.
|
||||
pub var dsdt_physical: u64 = 0;
|
||||
|
||||
@@ -165,8 +152,9 @@ var fadt_physical: u64 = 0;
|
||||
var fadt_length: u64 = 0;
|
||||
|
||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||
// for the sleep-state (`_Sx`) packages.
|
||||
// address + length of each table's post-header bytecode. The kernel does not
|
||||
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
|
||||
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
|
||||
var aml_block_physical: [32]u64 = undefined;
|
||||
var aml_block_len: [32]usize = undefined;
|
||||
var aml_block_count: usize = 0;
|
||||
@@ -373,7 +361,7 @@ const Hpet = extern struct {
|
||||
|
||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
|
||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
if (rsdp_physical == 0) return error.NoRsdp;
|
||||
boot_memory_regions = memory_regions;
|
||||
@@ -383,8 +371,6 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
||||
fadt_physical = 0;
|
||||
fadt_length = 0;
|
||||
platform_information = .{};
|
||||
aml_stats = .{};
|
||||
namespace = null;
|
||||
dsdt_physical = 0;
|
||||
aml_block_count = 0;
|
||||
|
||||
@@ -401,34 +387,20 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||
}
|
||||
|
||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
||||
// read the sleep types from it.
|
||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||
for (0..aml_block_count) |i| {
|
||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||
}
|
||||
const active = blocks[0..aml_block_count];
|
||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||
namespace = pr.namespace;
|
||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||
// The namespace's Device objects are no longer folded into the kernel
|
||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||
// (published below), re-parses the same blobs, and registers + reports
|
||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||
// \_S5 sleep type above.
|
||||
} else |_| {
|
||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||
// whatever the FADT alone provided.
|
||||
}
|
||||
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
|
||||
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
|
||||
// and the power register map. The AML bytecode (device enumeration and the
|
||||
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
|
||||
// job: it claims the acpi-tables node published below, parses the same blobs,
|
||||
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
|
||||
// AML in the kernel keeps boot latency off the critical, single-core path.
|
||||
|
||||
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||
// claimant. Kept even when the kernel-side device building (above) retires
|
||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
|
||||
// now that the kernel keeps only the *static* tables for itself.
|
||||
publishAcpiTablesNode(device_tree) catch {};
|
||||
}
|
||||
|
||||
@@ -461,12 +433,6 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||
}
|
||||
|
||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||
pub fn amlDeviceCount() usize {
|
||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
@@ -502,7 +468,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||
parseDmar(hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||
addAmlBlock(sdt_physical);
|
||||
}
|
||||
// Any other signature is recognised but left opaque for now.
|
||||
@@ -740,8 +706,8 @@ const fadt_x_pm_tmr_blk = 208; // GAS
|
||||
const flag_reset_register_supported = 1 << 10;
|
||||
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||
|
||||
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
||||
/// FADT -> the power register map (into `power_information`) and the DSDT address,
|
||||
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
|
||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const len: usize = header.length;
|
||||
|
||||
@@ -22,13 +22,14 @@ pub const Resource = device_model.Resource;
|
||||
pub const ResourceKind = device_model.ResourceKind;
|
||||
pub const Hal = device_model.Hal;
|
||||
pub const PowerInformation = acpi.PowerInformation;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInformation = acpi.PlatformInformation;
|
||||
pub const RegisterAccess = acpi.RegisterAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||
/// owned by the ring-3 acpi service), so they are not here.
|
||||
pub fn powerInformation() PowerInformation {
|
||||
return acpi.power_information;
|
||||
}
|
||||
@@ -39,18 +40,6 @@ pub fn platformInformation() PlatformInformation {
|
||||
return acpi.platform_information;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
||||
/// count against this.
|
||||
pub fn amlDeviceCount() usize {
|
||||
return acpi.amlDeviceCount();
|
||||
}
|
||||
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// The usable logical processors discovered during enumeration — one entry per
|
||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||
@@ -92,12 +81,8 @@ pub fn discover(
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls. Soft-off
|
||||
/// (S5) is not a kernel operation — the ring-3 acpi service owns it (docs/power.md).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
|
||||
+10
-63
@@ -1,34 +1,17 @@
|
||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
||||
//! Machine reboot: restart via the FADT reset register, with legacy fallbacks.
|
||||
//!
|
||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
||||
//!
|
||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
||||
//! re-initialisation, a milestone of its own.
|
||||
//! Built on the register map `acpi` extracted from the FADT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Soft-off (ACPI S5) and suspend (S3) are
|
||||
//! **not** here: they need the AML sleep-state (`_Sx`) values, which the kernel no
|
||||
//! longer parses — the ring-3 acpi service owns power management (it re-parses the
|
||||
//! blobs and writes the PM1 control register itself). See docs/power.md. Reboot
|
||||
//! stays in the kernel because it needs no AML — only the FADT reset register and
|
||||
//! the well-known legacy fallbacks — so it survives as a last-resort restart.
|
||||
|
||||
const acpi = @import("acpi.zig");
|
||||
const device_model = @import("device-model.zig");
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||
|
||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||
pub fn enable(hal: Hal) void {
|
||||
const pi = acpi.power_information;
|
||||
if (!pi.pm1a_cnt.present()) return;
|
||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||
|
||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||
var spins: usize = 0;
|
||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
}
|
||||
|
||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
@@ -48,42 +31,6 @@ pub fn reboot(hal: Hal) void {
|
||||
delay();
|
||||
}
|
||||
|
||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
enable(hal);
|
||||
const pi = acpi.power_information;
|
||||
const s5 = pi.s5 orelse return;
|
||||
|
||||
if (pi.pm1a_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
}
|
||||
if (pi.pm1b_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
}
|
||||
delay();
|
||||
}
|
||||
|
||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
||||
_ = hal;
|
||||
return error.Unsupported;
|
||||
}
|
||||
|
||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
||||
/// [12:10], SLP_EN in bit 13.
|
||||
fn sleepValue(slp_typ: u8) u32 {
|
||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||
}
|
||||
|
||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
return p.*;
|
||||
}
|
||||
return hal.pioRead(register.width, @intCast(register.address));
|
||||
}
|
||||
|
||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
@@ -93,8 +40,8 @@ fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// A short busy-wait so a reset takes effect before we fall through to the next
|
||||
/// method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// from being optimised away.
|
||||
fn delay() void {
|
||||
var i: usize = 0;
|
||||
|
||||
@@ -15,6 +15,7 @@ const op_inquiry: u8 = 0x12;
|
||||
const op_read_capacity_10: u8 = 0x25;
|
||||
const op_read_10: u8 = 0x28;
|
||||
const op_write_10: u8 = 0x2A;
|
||||
const op_synchronize_cache_10: u8 = 0x35;
|
||||
|
||||
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
||||
/// and product strings).
|
||||
@@ -57,6 +58,16 @@ pub fn write10(lba: u32, blocks: u16) [10]u8 {
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE(10): commit the device's write cache to stable media. LBA 0
|
||||
/// and block count 0 mean "the whole medium". No data stage. Without this a write
|
||||
/// can sit in the USB flash controller's cache and be lost if power is cut right
|
||||
/// after — which is exactly what a shutdown-time log flush hits on real hardware.
|
||||
pub fn synchronizeCache10() [10]u8 {
|
||||
var cdb = [_]u8{0} ** 10;
|
||||
cdb[0] = op_synchronize_cache_10;
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// Decode an 8-byte READ CAPACITY(10) reply.
|
||||
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
||||
return .{
|
||||
|
||||
@@ -147,6 +147,14 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.flush) => {
|
||||
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||
// cuts power. A device without a volatile cache reports success anyway.
|
||||
const cdb = scsi.synchronizeCache10();
|
||||
const ok = transact(&cdb, false, 0, 0);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -106,6 +106,12 @@ pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Whether a working UART was detected (loopback probe). When false the serial
|
||||
/// sink is silently inert — a dead legacy COM1 costs nothing per byte.
|
||||
pub fn serialPresent() bool {
|
||||
return serial.present();
|
||||
}
|
||||
|
||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||
/// displays. The last-resort progress signal when there's no text output at all.
|
||||
|
||||
@@ -7,8 +7,13 @@
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
//! The physmap (the permanent window onto all physical RAM) is built with 2 MiB
|
||||
//! huge pages wherever the range is 2 MiB-aligned, falling back to 4 KiB for the
|
||||
//! unaligned edges. On a big machine that is the difference between ~16.7M page-
|
||||
//! table entries (128 MiB of tables) and ~32K — it makes both the build and the
|
||||
//! footprint scale sanely with RAM. Everything else (kernel segments, heap, user
|
||||
//! space, on-demand MMIO) stays 4 KiB: precise, and the table memory is
|
||||
//! negligible there.
|
||||
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
@@ -22,10 +27,22 @@ const writable: u64 = 1 << 1;
|
||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||
const pwt: u64 = 1 << 3; // page write-through
|
||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||
const page_size_bit: u64 = 1 << 7; // PS: this PDPT/PD entry is a 1 GiB/2 MiB leaf, not a pointer to the next table
|
||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// The PAT-index bit. In a 4 KiB PTE it is bit 7; in a huge leaf (2 MiB PDE / 1 GiB
|
||||
// PDPTE) bit 7 is PS, so the PAT bit moves to bit 12. With PCD=PWT=0 this selects
|
||||
// PAT entry 4, which `setupPat` programs to write-combining (see mapRangePhysmap).
|
||||
const pte_pat: u64 = 1 << 7;
|
||||
const huge_pat: u64 = 1 << 12;
|
||||
const ia32_pat: u32 = 0x277;
|
||||
|
||||
/// The physmap's page size for 2 MiB-aligned RAM: one PD leaf covers this instead
|
||||
/// of 512 PT entries. 4 KiB pages fill the unaligned edges (see mapRangePhysmap).
|
||||
const huge_page_size: u64 = 2 << 20; // 2 MiB
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
@@ -74,7 +91,15 @@ fn allocTable() u64 {
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & address_mask;
|
||||
if (entry.* & present != 0) {
|
||||
// A present-but-huge entry is a leaf, not a table: descending would read
|
||||
// its 2 MiB/1 GiB data frame as a page table and corrupt RAM. This only
|
||||
// fires on a bug — a 4 KiB map landing inside a physmap huge page — and a
|
||||
// loud panic beats silent corruption. (The physmap and the 4 KiB regions
|
||||
// live in disjoint PML4 slots, so it should never happen.)
|
||||
if (entry.* & page_size_bit != 0) @panic("paging: descend through a huge-page leaf");
|
||||
return entry.* & address_mask;
|
||||
}
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
@@ -96,15 +121,39 @@ fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Map one 2 MiB huge page `virtual` -> `physical` with `flags` — a leaf at the PD
|
||||
/// level (PS bit set), with no PT beneath it. Both addresses must be 2 MiB-aligned.
|
||||
/// One of these replaces 512 `mapPage`s (and the PT frame they'd need).
|
||||
fn mapHugePage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||
@panic("paging: new higher-half PML4 entry after init");
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
tableAt(pd)[(virtual >> 21) & 0x1FF] = (physical & address_mask) | flags | present | page_size_bit;
|
||||
}
|
||||
|
||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||
/// window onto physical memory once the low identity map goes away.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||
/// window onto physical memory once the low identity map goes away. The 2 MiB-
|
||||
/// aligned interior is mapped with huge pages; the unaligned head/tail with 4 KiB.
|
||||
/// `write_combining` selects the WC memory type (setupPat's PAT entry 4) via the
|
||||
/// PAT bit — bit 7 in a 4 KiB PTE, bit 12 in a huge leaf — for the framebuffer.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64, write_combining: bool) void {
|
||||
const pte_flags = if (write_combining) flags | pte_pat else flags;
|
||||
const huge_flags = if (write_combining) flags | huge_pat else flags;
|
||||
var address = physical_base & ~@as(u64, page_size - 1);
|
||||
const end = physical_base + len;
|
||||
while (address < end) : (address += page_size) {
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
||||
}
|
||||
// Head: 4 KiB pages up to the next 2 MiB boundary.
|
||||
while (address < end and address & (huge_page_size - 1) != 0) : (address += page_size)
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||
// Interior: 2 MiB huge pages while a whole one still fits.
|
||||
while (address + huge_page_size <= end) : (address += huge_page_size)
|
||||
mapHugePage(pml4, boot_handoff.physicalToVirtual(address), address, huge_flags);
|
||||
// Tail: 4 KiB pages for whatever is left.
|
||||
while (address < end) : (address += page_size)
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||
}
|
||||
|
||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||
@@ -118,11 +167,26 @@ fn enableNx() void {
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Program this core's PAT so entry 4 (selected by the PAT bit with PCD=PWT=0) is
|
||||
/// **write-combining**, leaving the other seven at their reset types. Nothing else
|
||||
/// in danos sets the PAT bit, so this changes no existing mapping — it only gives
|
||||
/// the framebuffer a write-combining type, which turns its full-screen clear from
|
||||
/// glacial (uncached writes to a GPU BAR, the real-hardware default via MTRRs) into
|
||||
/// a batched burst. Must run on **every** core (PAT is per-logical-processor) — the
|
||||
/// framebuffer mapping lives in the shared kernel half, so a core with the reset
|
||||
/// PAT would see it as write-back and alias. Called from `init` (BSP) and each AP.
|
||||
pub fn setupPat() void {
|
||||
// Reset PAT is PA0=WB PA1=WT PA2=UC- PA3=UC PA4=WB PA5=WT PA6=UC- PA7=UC; flip
|
||||
// PA4 from WB (0x06) to WC (0x01). Type codes: UC=0 WC=1 WT=4 WP=5 WB=6 UC-=7.
|
||||
io.wrmsr(ia32_pat, 0x0007_0401_0007_0406);
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||
alloc_frame = allocFrame;
|
||||
free_frame = freeFrame;
|
||||
enableNx();
|
||||
setupPat(); // BSP: PAT entry 4 = write-combining, for the framebuffer window
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||
@@ -130,13 +194,15 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
||||
// mapped on demand (mapMmio) or explicitly below.
|
||||
for (regions(boot_information.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute, false);
|
||||
}
|
||||
|
||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||
// the kernel touches directly), RW + NX.
|
||||
// the kernel touches directly), RW + NX. The framebuffer is **write-
|
||||
// combining** (see setupPat) so the console's full-screen clear is a burst,
|
||||
// not millions of uncached single-word writes.
|
||||
const fb = boot_information.framebuffer;
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute, true);
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||
@@ -338,8 +404,8 @@ fn freeSubtree(physical: u64, level: u32) void {
|
||||
}
|
||||
|
||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||
/// clear. Walks the 4-level tables, stopping at a 2 MiB huge-page leaf (the physmap
|
||||
/// uses them). Returns false if unmapped. Used for W^X checks in tests.
|
||||
pub fn isExecutable(virtual: u64) bool {
|
||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return false;
|
||||
@@ -347,6 +413,7 @@ pub fn isExecutable(virtual: u64) bool {
|
||||
if (pdpte & present == 0) return false;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return false;
|
||||
if (pde & page_size_bit != 0) return pde & no_execute == 0; // 2 MiB huge leaf
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return false;
|
||||
return pte & no_execute == 0;
|
||||
@@ -385,9 +452,9 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||
/// needs the frame behind a user vaddr to free it).
|
||||
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
@@ -395,6 +462,8 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
if (pdpte & present == 0) return null;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return null;
|
||||
if (pde & page_size_bit != 0) // 2 MiB huge leaf: frame base is bits 51:21
|
||||
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return null;
|
||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||
|
||||
@@ -17,6 +17,12 @@ const Access = enum { port, mmio };
|
||||
var access: Access = .port;
|
||||
var base: u64 = 0x3F8; // COM1
|
||||
|
||||
/// Whether `init`/`reconfigure` found a *working* UART at `base`. False on a
|
||||
/// legacy-free machine whose COM1 is decoded but dead: writing to it is then a
|
||||
/// no-op, so `write` never spins waiting for a transmit register that will never
|
||||
/// drain. Cleared until proven by the loopback probe.
|
||||
var uart_present: bool = false;
|
||||
|
||||
fn portOut(p: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[p]"
|
||||
:
|
||||
@@ -57,6 +63,34 @@ pub fn init() void {
|
||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||
setRegister(4, 0x0B); // RTS/DSR set
|
||||
uart_present = probe();
|
||||
}
|
||||
|
||||
/// Detect a *working* UART by internal loopback: route the transmitter back to
|
||||
/// the receiver (MCR bit 4), send a byte, and check it comes back. A port that is
|
||||
/// merely decoded but has nothing behind it (the common case on a legacy-free
|
||||
/// board that still answers I/O at 0x3F8) never echoes, so this returns false.
|
||||
///
|
||||
/// This matters for speed, not just correctness: a dead UART's line-status
|
||||
/// register reads back 0x00, so its transmit-holding-empty bit never sets, and
|
||||
/// `writeByte` would otherwise spin its full guard — tens of milliseconds — on
|
||||
/// *every* logged byte. On real hardware that alone can add ~a minute to boot.
|
||||
fn probe() bool {
|
||||
const saved_mcr = register(4);
|
||||
setRegister(4, 0x1E); // MCR: LOOP | OUT2 | OUT1 | RTS — internal loopback
|
||||
setRegister(0, 0xAE); // push a distinctive byte into the loopback path
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x01 == 0 and guard < 10_000) : (guard += 1) {} // await Data Ready
|
||||
const echo = register(0);
|
||||
setRegister(4, saved_mcr); // restore the modem-control lines
|
||||
return echo == 0xAE;
|
||||
}
|
||||
|
||||
/// Whether a working UART was detected (see `probe`). The log sink stays
|
||||
/// registered regardless — it simply does nothing until this is true — so a UART
|
||||
/// that only `reconfigure` discovers (via SPCR) still starts logging.
|
||||
pub fn present() bool {
|
||||
return uart_present;
|
||||
}
|
||||
|
||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||
@@ -70,15 +104,19 @@ pub fn reconfigure(is_mmio: bool, address: u64) void {
|
||||
}
|
||||
|
||||
fn writeByte(c: u8) void {
|
||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
||||
// Wait for the transmit-holding register to empty. `write` only reaches here
|
||||
// for a UART the loopback probe proved live, so this bounds a momentary stall
|
||||
// (e.g. deasserted flow control), not an absent port: ~5000 legacy-port reads
|
||||
// is a few ms — comfortably longer than one 38400-baud byte-time (~260 µs).
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
||||
while (register(5) & 0x20 == 0 and guard < 5_000) : (guard += 1) {}
|
||||
setRegister(0, c);
|
||||
}
|
||||
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up. A no-op
|
||||
/// when no working UART was detected, so a dead COM1 costs nothing per byte.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
if (!uart_present) return;
|
||||
for (bytes) |c| {
|
||||
if (c == '\n') writeByte('\r');
|
||||
writeByte(c);
|
||||
|
||||
@@ -173,6 +173,7 @@ fn delayMicros(us: u64) void {
|
||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
const cpu = boot_index;
|
||||
paging.setupPat(); // this core's PAT: entry 4 = write-combining, to match the BSP
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
|
||||
+31
-16
@@ -60,8 +60,16 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||
// with port-0x80 checkpoints as the only progress signal.
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
//
|
||||
// Serial is compiled in only under -Dserial (build.zig): a real machine often
|
||||
// has no live legacy COM1, and the log survives in the RAM buffer (below) and
|
||||
// is flushed to disk — so serial is now a QEMU/dev convenience the flashable
|
||||
// image leaves out. When it *is* built in, `serialInit`'s loopback probe still
|
||||
// guards against a dead port (so a -Dserial image is safe on real hardware).
|
||||
if (build_options.serial) {
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
}
|
||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||
@@ -72,8 +80,14 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||
//
|
||||
// The console is brought up *after* paging (below), not here: its one-time
|
||||
// full-screen clear then runs on the kernel's **write-combining** mapping of the
|
||||
// framebuffer instead of the loader's uncached one — a fast burst rather than
|
||||
// millions of uncached writes on real hardware. Until then, on-screen output is
|
||||
// absent (an early panic still lands in the serial/RAM log); the trade is worth
|
||||
// a near-instant boot. `console.write` is a safe no-op while the console is down.
|
||||
const fb = boot_information.framebuffer;
|
||||
console.init(fb);
|
||||
|
||||
log.checkpoint(cp_entry);
|
||||
|
||||
@@ -83,10 +97,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
architecture.init();
|
||||
|
||||
status("/system/kernel: initialising kernel...\n");
|
||||
log.write(if (console.present())
|
||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
if (build_options.serial) log.write(if (architecture.serialPresent())
|
||||
"/system/kernel: serial console online (COM1)\n"
|
||||
else
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
"/system/kernel: no serial UART (COM1 absent) -> log kept in RAM/debugcon\n");
|
||||
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
@@ -142,6 +156,15 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Now on our own tables, the framebuffer window is write-combining: bring up
|
||||
// the on-screen console and clear it (a fast burst here, not the loader's
|
||||
// uncached crawl). From here `status` reaches the screen as well as the log.
|
||||
console.init(fb);
|
||||
log.write(if (console.present())
|
||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
else
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
log.checkpoint(cp_heap);
|
||||
@@ -176,21 +199,13 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||
const pw = platform.powerInformation();
|
||||
log.write("/system/kernel: power\n");
|
||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
} else {
|
||||
log.write(" S5 slp_typ : (not found)\n");
|
||||
}
|
||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||
|
||||
// AML namespace parse integrity: consumed should equal total.
|
||||
const am = platform.amlStats();
|
||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||
|
||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||
|
||||
+23
-34
@@ -181,10 +181,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
containmentTest();
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "poweroff")) {
|
||||
powerTest(.off);
|
||||
} else if (eql(case, "reboot")) {
|
||||
powerTest(.reboot);
|
||||
rebootTest();
|
||||
} else {
|
||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||
}
|
||||
@@ -201,15 +199,14 @@ fn platformHal() platform.Hal {
|
||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
||||
/// transition failed and we emit a FAIL result.
|
||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
||||
const name = if (action == .off) "poweroff" else "reboot";
|
||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
||||
// Soft-off (S5) is no longer a kernel operation — the ring-3 acpi service owns it
|
||||
// (exercised end-to-end by `orderly-shutdown`). Reboot stays in the kernel (FADT
|
||||
// reset register, no AML), so it keeps its own case.
|
||||
fn rebootTest() void {
|
||||
log("DANOS-TEST-BEGIN: reboot\n", .{});
|
||||
const hal = platformHal();
|
||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
||||
switch (action) {
|
||||
.off => platform.shutdown(hal),
|
||||
.reboot => platform.reboot(hal),
|
||||
}
|
||||
log("DANOS-POWER: attempting reboot\n", .{});
|
||||
platform.reboot(hal);
|
||||
check("power transition took effect", false);
|
||||
result();
|
||||
}
|
||||
@@ -276,22 +273,19 @@ fn timer() void {
|
||||
}
|
||||
|
||||
/// Verify device discovery populated the platform facts the rest of the kernel
|
||||
/// depends on — the results ACPI parsing stashed in globals at boot. These are
|
||||
/// stable for the QEMU q35 + OVMF machine the harness runs, and span the tables:
|
||||
/// MADT (LAPIC base, CPU count), FADT (PM/reset registers), and the AML parse
|
||||
/// (the sleep type, plus the integrity check that every byte was consumed).
|
||||
/// depends on — the results the static ACPI tables stashed in globals at boot.
|
||||
/// These are stable for the QEMU q35 + OVMF machine the harness runs, and span the
|
||||
/// tables: MADT (LAPIC base, CPU count) and FADT (PM/reset registers). The kernel
|
||||
/// no longer interprets AML — sleep types are the ring-3 acpi service's concern.
|
||||
fn discoveryTest() void {
|
||||
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
||||
const pinfo = platform.platformInformation();
|
||||
const pw = platform.powerInformation();
|
||||
const am = platform.amlStats();
|
||||
|
||||
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
||||
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
||||
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
||||
check("reset register supported (FADT)", pw.reset_supported);
|
||||
check("S5 sleep type found (AML)", pw.s5 != null);
|
||||
check("AML parsed completely (consumed == total)", am.total > 0 and am.consumed == am.total);
|
||||
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
||||
|
||||
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
||||
@@ -2095,11 +2089,11 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// M20.1: the ring-3 AML parse agrees with the kernel's. The manager spawns
|
||||
/// the discovery service (the acpi build variant); it claims the acpi-tables
|
||||
/// node, maps the blobs, parses them, and logs its Device count — which must
|
||||
/// equal what the kernel's own parse produced (the equivalence that licenses
|
||||
/// retiring the kernel's device build in M20.3).
|
||||
/// M20.1: the ring-3 AML parse works. The manager spawns the discovery service
|
||||
/// (the acpi build variant); it claims the acpi-tables node, maps the blobs,
|
||||
/// parses them, and self-verifies it found at least a floor of Device objects,
|
||||
/// printing "acpi-parse: ok". The kernel no longer parses AML, so there is no
|
||||
/// kernel count to compare against — the ring-3 parse is now the only one.
|
||||
fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
@@ -2114,23 +2108,18 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||
return;
|
||||
};
|
||||
|
||||
// The kernel's own count, from the namespace it already built for \_S5.
|
||||
const kernel_devices = platform.amlDeviceCount();
|
||||
check("the kernel namespace has devices to compare against", kernel_devices >= 1);
|
||||
|
||||
// Spawn the discovery service directly with that count as argv: it parses
|
||||
// the same blobs in ring 3 and self-verifies, printing "acpi-parse: ok" iff
|
||||
// the counts match. The harness's expect regex is that marker — deterministic,
|
||||
// no racing the shared serial buffer.
|
||||
// Spawn the discovery service directly with a device-count *floor* as argv:
|
||||
// it parses the blobs in ring 3 and self-verifies it found at least that many
|
||||
// Device objects, printing "acpi-parse: ok". A floor of 1 just proves the
|
||||
// parser ran and produced a namespace (the QEMU q35 DSDT has dozens). The
|
||||
// marker is deterministic — no racing the shared serial buffer.
|
||||
process.setInitialRamdisk(image);
|
||||
var count_text: [16]u8 = undefined;
|
||||
const count_arg = std.fmt.bufPrint(&count_text, "{d}", .{kernel_devices}) catch "0";
|
||||
var spawned = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "discovery")) continue;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", count_arg }, scheduler.currentId(), null) catch 0;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
|
||||
spawned = true;
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -103,9 +103,11 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
||||
// own device count to self-verify against — deterministic, no log-scraping.
|
||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||
// devices is the check. Deterministic, no log-scraping.
|
||||
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||
@@ -162,11 +164,11 @@ pub fn main(init: runtime.process.Init) void {
|
||||
var namespace = result.namespace;
|
||||
const devices = aml.deviceCount(&namespace);
|
||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||
if (expected) |want| {
|
||||
if (devices == want) {
|
||||
if (floor) |minimum| {
|
||||
if (devices >= minimum) {
|
||||
_ = runtime.system.write("acpi-parse: ok\n");
|
||||
} else {
|
||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
||||
writeLine("acpi-parse: too few (ring-3 {d} < floor {d})\n", .{ devices, minimum });
|
||||
}
|
||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
|
||||
@@ -16,6 +16,10 @@ pub const Operation = enum(u32) {
|
||||
read = 1,
|
||||
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||
write = 2,
|
||||
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
|
||||
@@ -40,11 +40,16 @@ const IpcBlock = struct {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(destination[0..512], buffer[0..512]);
|
||||
return self.device.write(lba, 1, self.bounce.physical);
|
||||
if (!self.device.write(lba, 1, self.bounce.physical)) return false;
|
||||
device_dirty = true; // a block reached the device; a close will flush it
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
var ipc_block: IpcBlock = undefined;
|
||||
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
|
||||
// Open handles the VFS holds against this backend: each maps a node id to a
|
||||
@@ -192,6 +197,14 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
||||
},
|
||||
.close => {
|
||||
if (openAt(request.node)) |o| o.used = false;
|
||||
// Durable-on-close: if any block reached the device since the last
|
||||
// flush, commit its cache to stable media now (best-effort). This is
|
||||
// what makes init's shutdown log flush survive a real power-off, and is
|
||||
// the right default for removable media the user may unplug.
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.mkdir => {
|
||||
|
||||
+7
-5
@@ -365,7 +365,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"acpi-parse: ok",
|
||||
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
||||
"fail": r"acpi-parse: too few|DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
||||
@@ -492,9 +492,8 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
||||
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
||||
{"name": "poweroff",
|
||||
"expect": r"DANOS-POWER: attempting poweroff",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Soft-off (S5) is owned by the ring-3 acpi service now (see orderly-shutdown);
|
||||
# the kernel keeps only reboot (FADT reset register, no AML).
|
||||
{"name": "reboot",
|
||||
"expect": r"DANOS-POWER: attempting reboot",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -504,7 +503,10 @@ TIMEOUT = 30 # seconds per case
|
||||
|
||||
|
||||
def build(arch, case):
|
||||
cmd = ["zig", "build", f"-Dtest-case={case}"] + arch["zig_flags"]
|
||||
# -Dserial: the harness asserts on markers the kernel writes to serial0, so the
|
||||
# serial log sink must be compiled in. It is off by default (a flashed real-
|
||||
# hardware image keeps its log in RAM instead; see build.zig / serial.zig).
|
||||
cmd = ["zig", "build", f"-Dtest-case={case}", "-Dserial=true"] + arch["zig_flags"]
|
||||
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
|
||||
if r.returncode != 0:
|
||||
return r.stderr.strip() or r.stdout.strip()
|
||||
|
||||
Reference in New Issue
Block a user