Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f |
+12
-3
@@ -2,6 +2,7 @@ const std = @import("std");
|
|||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const build_options = @import("build_options");
|
||||||
const BootInformation = boot_handoff.BootInformation;
|
const BootInformation = boot_handoff.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
@@ -84,7 +85,7 @@ fn boot() !noreturn {
|
|||||||
// the map and exiting would invalidate the map key.
|
// the map and exiting would invalidate the map key.
|
||||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
log("EFI: kernel loaded, exiting boot services\r\n");
|
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||||
boot_information.memory_map = try exitBootServices(bs);
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
@@ -395,7 +396,7 @@ fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !
|
|||||||
const image = try loadFile(bs, init_file_name);
|
const image = try loadFile(bs, init_file_name);
|
||||||
boot_information.init_base = @intFromPtr(image.ptr);
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
boot_information.init_len = image.len;
|
boot_information.init_len = image.len;
|
||||||
log("EFI: /system/services/init loaded\r\n");
|
progress("EFI: /system/services/init loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||||
@@ -403,7 +404,7 @@ fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInfo
|
|||||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||||
boot_information.initial_ramdisk_len = image.len;
|
boot_information.initial_ramdisk_len = image.len;
|
||||||
log("EFI: initial_ramdisk loaded\r\n");
|
progress("EFI: initial_ramdisk loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
@@ -561,6 +562,14 @@ fn log(comptime message: []const u8) void {
|
|||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||||
|
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||||
|
/// `log` directly and always show, so a failed boot still explains itself.
|
||||||
|
fn progress(comptime message: []const u8) void {
|
||||||
|
if (!build_options.serial) return;
|
||||||
|
log(message);
|
||||||
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
|
|||||||
@@ -96,6 +96,105 @@ fn addUserBinary(
|
|||||||
return exe;
|
return exe;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||||
|
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||||
|
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||||
|
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||||
|
const KernelModules = struct {
|
||||||
|
boot_handoff: *std.Build.Module,
|
||||||
|
abi: *std.Build.Module,
|
||||||
|
device_abi: *std.Build.Module,
|
||||||
|
architecture: *std.Build.Module,
|
||||||
|
platform: *std.Build.Module,
|
||||||
|
parameters: *std.Build.Module,
|
||||||
|
initial_ramdisk: *std.Build.Module,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||||
|
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||||
|
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||||
|
/// build option baked into `build_options`.
|
||||||
|
fn addKernel(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_target: std.Build.ResolvedTarget,
|
||||||
|
optimize: std.builtin.OptimizeMode,
|
||||||
|
modules: KernelModules,
|
||||||
|
test_case: ?[]const u8,
|
||||||
|
serial: bool,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||||
|
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||||
|
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||||
|
const build_options = b.addOptions();
|
||||||
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
|
build_options.addOption(bool, "serial", serial);
|
||||||
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = "kernel",
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||||
|
.target = kernel_target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
|
.stack_protector = false,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||||
|
.{ .name = "abi", .module = modules.abi },
|
||||||
|
.{ .name = "device-abi", .module = modules.device_abi },
|
||||||
|
.{ .name = "architecture", .module = modules.architecture },
|
||||||
|
.{ .name = "platform", .module = modules.platform },
|
||||||
|
.{ .name = "parameters", .module = modules.parameters },
|
||||||
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
|
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||||
|
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||||
|
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||||
|
/// bundle its own kernel while sharing the (serial-independent) loader, init, and
|
||||||
|
/// ramdisk. Returns the image's LazyPath.
|
||||||
|
fn addBootImage(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_bin: std.Build.LazyPath,
|
||||||
|
efi_bin: std.Build.LazyPath,
|
||||||
|
init_bin: std.Build.LazyPath,
|
||||||
|
initial_ramdisk_img: std.Build.LazyPath,
|
||||||
|
) std.Build.LazyPath {
|
||||||
|
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||||
|
mk_fat.addArg("64"); // MiB
|
||||||
|
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||||
|
mk_fat.addFileArg(efi_bin);
|
||||||
|
mk_fat.addArg("system/kernel");
|
||||||
|
mk_fat.addFileArg(kernel_bin);
|
||||||
|
mk_fat.addArg("system/services/init");
|
||||||
|
mk_fat.addFileArg(init_bin);
|
||||||
|
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||||
|
mk_fat.addFileArg(initial_ramdisk_img);
|
||||||
|
return fat_image;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
@@ -285,9 +384,12 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
// The serial-console log sink. Off by default: a real machine often has no
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||||
const build_options_module = build_options.createModule();
|
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||||
|
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||||
|
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||||
|
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -299,41 +401,17 @@ pub fn build(b: *std.Build) void {
|
|||||||
.abi = .none,
|
.abi = .none,
|
||||||
});
|
});
|
||||||
|
|
||||||
const exe = b.addExecutable(.{
|
const kernel_modules = KernelModules{
|
||||||
.name = "kernel",
|
.boot_handoff = boot_handoff_module,
|
||||||
.root_module = b.createModule(.{
|
.abi = abi_module,
|
||||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
.device_abi = device_abi_module,
|
||||||
.target = kernel_target,
|
.architecture = architecture_module,
|
||||||
.optimize = optimize,
|
.platform = platform_module,
|
||||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
.parameters = parameters_module,
|
||||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
.initial_ramdisk = initial_ramdisk_module,
|
||||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
};
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||||
.stack_protector = false,
|
|
||||||
.imports = &.{
|
|
||||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
|
||||||
.{ .name = "abi", .module = abi_module },
|
|
||||||
.{ .name = "device-abi", .module = device_abi_module },
|
|
||||||
.{ .name = "architecture", .module = architecture_module },
|
|
||||||
.{ .name = "platform", .module = platform_module },
|
|
||||||
.{ .name = "parameters", .module = parameters_module },
|
|
||||||
.{ .name = "build_options", .module = build_options_module },
|
|
||||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
|
||||||
},
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
|
||||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
|
||||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
|
||||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
|
||||||
exe.use_llvm = true;
|
|
||||||
exe.use_lld = true;
|
|
||||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
|
||||||
// linker's AT() clauses give each segment a low physical load address
|
|
||||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
|
||||||
exe.image_base = 0xFFFFFFFF80100000;
|
|
||||||
|
|
||||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
@@ -498,6 +576,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||||
// Each is its own binary/entry (a loader is built for its own target); today
|
// Each is its own binary/entry (a loader is built for its own target); today
|
||||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||||
|
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||||
|
// which firmware may mirror to a serial console) are silenced by default — a
|
||||||
|
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||||
|
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||||
|
const loader_options = b.addOptions();
|
||||||
|
loader_options.addOption(bool, "serial", serial);
|
||||||
|
const loader_options_module = loader_options.createModule();
|
||||||
const efiexe = b.addExecutable(.{
|
const efiexe = b.addExecutable(.{
|
||||||
.name = "BOOTX64",
|
.name = "BOOTX64",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
@@ -510,6 +595,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.imports = &.{
|
.imports = &.{
|
||||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
|
.{ .name = "build_options", .module = loader_options_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
@@ -525,21 +611,17 @@ pub fn build(b: *std.Build) void {
|
|||||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
|
||||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
|
||||||
mk_fat.addArg("64"); // MiB
|
|
||||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
|
||||||
mk_fat.addFileArg(efiexe.getEmittedBin());
|
|
||||||
mk_fat.addArg("system/kernel");
|
|
||||||
mk_fat.addFileArg(exe.getEmittedBin());
|
|
||||||
mk_fat.addArg("system/services/init");
|
|
||||||
mk_fat.addFileArg(init_exe.getEmittedBin());
|
|
||||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
|
||||||
mk_fat.addFileArg(initial_ramdisk_img);
|
|
||||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|
||||||
|
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||||
|
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||||
|
// log captured to serial0 — without baking serial into the image users flash.
|
||||||
|
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||||
|
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||||
|
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||||
|
|
||||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||||
@@ -611,8 +693,9 @@ pub fn build(b: *std.Build) void {
|
|||||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||||
|
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||||
run_efi.addArg("-drive");
|
run_efi.addArg("-drive");
|
||||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image);
|
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||||
run_efi.addArgs(&.{
|
run_efi.addArgs(&.{
|
||||||
"-device",
|
"-device",
|
||||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||||
@@ -634,8 +717,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||||
run_efi.step.dependOn(b.getInstallStep());
|
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||||
|
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||||
|
// scratch dir first.
|
||||||
run_efi.step.dependOn(&make_log_dir.step);
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
|
|||||||
@@ -27,6 +27,15 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
|||||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||||
a new architecture's UART is what makes the same tests run there.
|
a new architecture's UART is what makes the same tests run there.
|
||||||
|
|
||||||
|
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||||
|
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||||
|
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||||
|
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||||
|
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||||
|
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||||
|
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||||
|
so a serial-enabled image is still safe on real hardware.)
|
||||||
|
|
||||||
## In-kernel test cases
|
## In-kernel test cases
|
||||||
|
|
||||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||||
|
|||||||
@@ -38,6 +38,13 @@ pub const Device = struct {
|
|||||||
return self.transfer(.write, lba, count, physical);
|
return self.transfer(.write, lba, count, physical);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||||
|
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||||
|
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||||
|
pub fn flush(self: Device) bool {
|
||||||
|
return self.transfer(.flush, 0, 0, 0);
|
||||||
|
}
|
||||||
|
|
||||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||||
var reply: [protocol.reply_size]u8 = undefined;
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
|||||||
+28
-62
@@ -3,11 +3,13 @@
|
|||||||
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||||
//! bootloader handed us) and translates the static tables into the generic
|
//! bootloader handed us) and translates the static tables into the generic
|
||||||
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||||
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
//! source. This is deliberately the *static-table* path, and **only** that: MADT
|
||||||
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
|
||||||
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
|
||||||
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
//! interpreted here — the kernel collects the blobs and publishes them on the
|
||||||
//! interpretation is a separate, larger subproject.
|
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
|
||||||
|
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
|
||||||
|
//! off the single-core critical path (nothing else runs alongside it there).
|
||||||
//!
|
//!
|
||||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||||
@@ -19,7 +21,6 @@ const boot_handoff = @import("boot-handoff");
|
|||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const parameters = @import("parameters");
|
const parameters = @import("parameters");
|
||||||
const device_model = @import("device-model.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const aml = @import("aml/aml.zig");
|
|
||||||
const DeviceTree = device_model.DeviceTree;
|
const DeviceTree = device_model.DeviceTree;
|
||||||
const Hal = device_model.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
@@ -37,8 +38,11 @@ pub const RegisterAccess = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
/// The power register map, extracted from the FADT during discovery. Populated by
|
||||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
|
||||||
|
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
|
||||||
|
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
|
||||||
|
/// writes the PM1 control register itself). So this holds only the FADT scalars.
|
||||||
pub const PowerInformation = struct {
|
pub const PowerInformation = struct {
|
||||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||||
@@ -54,10 +58,6 @@ pub const PowerInformation = struct {
|
|||||||
reset: RegisterAccess = .{},
|
reset: RegisterAccess = .{},
|
||||||
reset_value: u8 = 0,
|
reset_value: u8 = 0,
|
||||||
reset_supported: bool = false,
|
reset_supported: bool = false,
|
||||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
|
||||||
/// packages.
|
|
||||||
s5: ?aml.SleepType = null,
|
|
||||||
s3: ?aml.SleepType = null,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||||
@@ -141,19 +141,6 @@ const maximum_cpus = parameters.maximum_cpus;
|
|||||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||||
pub var cpu_information: CpuInformation = .{};
|
pub var cpu_information: CpuInformation = .{};
|
||||||
|
|
||||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
|
||||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
|
||||||
pub const AmlStats = struct {
|
|
||||||
nodes: usize = 0,
|
|
||||||
consumed: usize = 0,
|
|
||||||
total: usize = 0,
|
|
||||||
};
|
|
||||||
pub var aml_stats: AmlStats = .{};
|
|
||||||
|
|
||||||
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
|
||||||
/// device enumeration later. Null until `discover` runs successfully.
|
|
||||||
pub var namespace: ?aml.Namespace = null;
|
|
||||||
|
|
||||||
/// Physical address of the DSDT the FADT points at, or 0.
|
/// Physical address of the DSDT the FADT points at, or 0.
|
||||||
pub var dsdt_physical: u64 = 0;
|
pub var dsdt_physical: u64 = 0;
|
||||||
|
|
||||||
@@ -165,8 +152,9 @@ var fadt_physical: u64 = 0;
|
|||||||
var fadt_length: u64 = 0;
|
var fadt_length: u64 = 0;
|
||||||
|
|
||||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
// address + length of each table's post-header bytecode. The kernel does not
|
||||||
// for the sleep-state (`_Sx`) packages.
|
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
|
||||||
|
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
|
||||||
var aml_block_physical: [32]u64 = undefined;
|
var aml_block_physical: [32]u64 = undefined;
|
||||||
var aml_block_len: [32]usize = undefined;
|
var aml_block_len: [32]usize = undefined;
|
||||||
var aml_block_count: usize = 0;
|
var aml_block_count: usize = 0;
|
||||||
@@ -373,7 +361,7 @@ const Hpet = extern struct {
|
|||||||
|
|
||||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
|
||||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
if (rsdp_physical == 0) return error.NoRsdp;
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
boot_memory_regions = memory_regions;
|
boot_memory_regions = memory_regions;
|
||||||
@@ -383,8 +371,6 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
fadt_physical = 0;
|
fadt_physical = 0;
|
||||||
fadt_length = 0;
|
fadt_length = 0;
|
||||||
platform_information = .{};
|
platform_information = .{};
|
||||||
aml_stats = .{};
|
|
||||||
namespace = null;
|
|
||||||
dsdt_physical = 0;
|
dsdt_physical = 0;
|
||||||
aml_block_count = 0;
|
aml_block_count = 0;
|
||||||
|
|
||||||
@@ -401,34 +387,20 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
|
||||||
// read the sleep types from it.
|
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
|
||||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
// and the power register map. The AML bytecode (device enumeration and the
|
||||||
for (0..aml_block_count) |i| {
|
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
|
||||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
// job: it claims the acpi-tables node published below, parses the same blobs,
|
||||||
}
|
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
|
||||||
const active = blocks[0..aml_block_count];
|
// AML in the kernel keeps boot latency off the critical, single-core path.
|
||||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
|
||||||
namespace = pr.namespace;
|
|
||||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
|
||||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
|
||||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
|
||||||
// The namespace's Device objects are no longer folded into the kernel
|
|
||||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
|
||||||
// (published below), re-parses the same blobs, and registers + reports
|
|
||||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
|
||||||
// \_S5 sleep type above.
|
|
||||||
} else |_| {
|
|
||||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
|
||||||
// whatever the FADT alone provided.
|
|
||||||
}
|
|
||||||
|
|
||||||
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
// claimant. Kept even when the kernel-side device building (above) retires
|
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
|
||||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
// now that the kernel keeps only the *static* tables for itself.
|
||||||
publishAcpiTablesNode(device_tree) catch {};
|
publishAcpiTablesNode(device_tree) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -461,12 +433,6 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
|||||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
|
||||||
pub fn amlDeviceCount() usize {
|
|
||||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
@@ -502,7 +468,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
|||||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||||
parseDmar(hal, header);
|
parseDmar(hal, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||||
addAmlBlock(sdt_physical);
|
addAmlBlock(sdt_physical);
|
||||||
}
|
}
|
||||||
// Any other signature is recognised but left opaque for now.
|
// Any other signature is recognised but left opaque for now.
|
||||||
@@ -740,8 +706,8 @@ const fadt_x_pm_tmr_blk = 208; // GAS
|
|||||||
const flag_reset_register_supported = 1 << 10;
|
const flag_reset_register_supported = 1 << 10;
|
||||||
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||||
|
|
||||||
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
/// FADT -> the power register map (into `power_information`) and the DSDT address,
|
||||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
|
||||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
const len: usize = header.length;
|
const len: usize = header.length;
|
||||||
|
|||||||
@@ -22,13 +22,14 @@ pub const Resource = device_model.Resource;
|
|||||||
pub const ResourceKind = device_model.ResourceKind;
|
pub const ResourceKind = device_model.ResourceKind;
|
||||||
pub const Hal = device_model.Hal;
|
pub const Hal = device_model.Hal;
|
||||||
pub const PowerInformation = acpi.PowerInformation;
|
pub const PowerInformation = acpi.PowerInformation;
|
||||||
pub const AmlStats = acpi.AmlStats;
|
|
||||||
pub const PlatformInformation = acpi.PlatformInformation;
|
pub const PlatformInformation = acpi.PlatformInformation;
|
||||||
pub const RegisterAccess = acpi.RegisterAccess;
|
pub const RegisterAccess = acpi.RegisterAccess;
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
pub const IsoEntry = acpi.IsoEntry;
|
||||||
pub const Cpu = acpi.Cpu;
|
pub const Cpu = acpi.Cpu;
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||||
|
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||||
|
/// owned by the ring-3 acpi service), so they are not here.
|
||||||
pub fn powerInformation() PowerInformation {
|
pub fn powerInformation() PowerInformation {
|
||||||
return acpi.power_information;
|
return acpi.power_information;
|
||||||
}
|
}
|
||||||
@@ -39,18 +40,6 @@ pub fn platformInformation() PlatformInformation {
|
|||||||
return acpi.platform_information;
|
return acpi.platform_information;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
|
||||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
|
||||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
|
||||||
/// count against this.
|
|
||||||
pub fn amlDeviceCount() usize {
|
|
||||||
return acpi.amlDeviceCount();
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn amlStats() AmlStats {
|
|
||||||
return acpi.aml_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The usable logical processors discovered during enumeration — one entry per
|
/// The usable logical processors discovered during enumeration — one entry per
|
||||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||||
@@ -92,12 +81,8 @@ pub fn discover(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls. Soft-off
|
||||||
|
/// (S5) is not a kernel operation — the ring-3 acpi service owns it (docs/power.md).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
power.reboot(hal);
|
power.reboot(hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Power the machine off (ACPI S5). Never returns on success.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
power.shutdown(hal);
|
|
||||||
}
|
|
||||||
|
|||||||
+10
-63
@@ -1,34 +1,17 @@
|
|||||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
//! Machine reboot: restart via the FADT reset register, with legacy fallbacks.
|
||||||
//!
|
//!
|
||||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
//! Built on the register map `acpi` extracted from the FADT, driven through the
|
||||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
//! injected `Hal` (port I/O and MMIO). Soft-off (ACPI S5) and suspend (S3) are
|
||||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
//! **not** here: they need the AML sleep-state (`_Sx`) values, which the kernel no
|
||||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
//! longer parses — the ring-3 acpi service owns power management (it re-parses the
|
||||||
//!
|
//! blobs and writes the PM1 control register itself). See docs/power.md. Reboot
|
||||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
//! stays in the kernel because it needs no AML — only the FADT reset register and
|
||||||
//! re-initialisation, a milestone of its own.
|
//! the well-known legacy fallbacks — so it survives as a last-resort restart.
|
||||||
|
|
||||||
const acpi = @import("acpi.zig");
|
const acpi = @import("acpi.zig");
|
||||||
const device_model = @import("device-model.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const Hal = device_model.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
|
||||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
|
||||||
|
|
||||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
|
||||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
|
||||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
|
||||||
pub fn enable(hal: Hal) void {
|
|
||||||
const pi = acpi.power_information;
|
|
||||||
if (!pi.pm1a_cnt.present()) return;
|
|
||||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
|
||||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
|
||||||
|
|
||||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
|
||||||
var spins: usize = 0;
|
|
||||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
@@ -48,42 +31,6 @@ pub fn reboot(hal: Hal) void {
|
|||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
|
||||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
enable(hal);
|
|
||||||
const pi = acpi.power_information;
|
|
||||||
const s5 = pi.s5 orelse return;
|
|
||||||
|
|
||||||
if (pi.pm1a_cnt.present()) {
|
|
||||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
|
||||||
}
|
|
||||||
if (pi.pm1b_cnt.present()) {
|
|
||||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
|
||||||
}
|
|
||||||
delay();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
|
||||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
|
||||||
_ = hal;
|
|
||||||
return error.Unsupported;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
|
||||||
/// [12:10], SLP_EN in bit 13.
|
|
||||||
fn sleepValue(slp_typ: u8) u32 {
|
|
||||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
|
||||||
if (register.mmio) {
|
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
|
||||||
return p.*;
|
|
||||||
}
|
|
||||||
return hal.pioRead(register.width, @intCast(register.address));
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||||
if (register.mmio) {
|
if (register.mmio) {
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
@@ -93,8 +40,8 @@ fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
/// A short busy-wait so a reset takes effect before we fall through to the next
|
||||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
/// method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||||
/// from being optimised away.
|
/// from being optimised away.
|
||||||
fn delay() void {
|
fn delay() void {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ const op_inquiry: u8 = 0x12;
|
|||||||
const op_read_capacity_10: u8 = 0x25;
|
const op_read_capacity_10: u8 = 0x25;
|
||||||
const op_read_10: u8 = 0x28;
|
const op_read_10: u8 = 0x28;
|
||||||
const op_write_10: u8 = 0x2A;
|
const op_write_10: u8 = 0x2A;
|
||||||
|
const op_synchronize_cache_10: u8 = 0x35;
|
||||||
|
|
||||||
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
||||||
/// and product strings).
|
/// and product strings).
|
||||||
@@ -57,6 +58,16 @@ pub fn write10(lba: u32, blocks: u16) [10]u8 {
|
|||||||
return cdb;
|
return cdb;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// SYNCHRONIZE CACHE(10): commit the device's write cache to stable media. LBA 0
|
||||||
|
/// and block count 0 mean "the whole medium". No data stage. Without this a write
|
||||||
|
/// can sit in the USB flash controller's cache and be lost if power is cut right
|
||||||
|
/// after — which is exactly what a shutdown-time log flush hits on real hardware.
|
||||||
|
pub fn synchronizeCache10() [10]u8 {
|
||||||
|
var cdb = [_]u8{0} ** 10;
|
||||||
|
cdb[0] = op_synchronize_cache_10;
|
||||||
|
return cdb;
|
||||||
|
}
|
||||||
|
|
||||||
/// Decode an 8-byte READ CAPACITY(10) reply.
|
/// Decode an 8-byte READ CAPACITY(10) reply.
|
||||||
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
||||||
return .{
|
return .{
|
||||||
|
|||||||
@@ -147,6 +147,14 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
|||||||
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||||
},
|
},
|
||||||
|
@intFromEnum(block_protocol.Operation.flush) => {
|
||||||
|
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||||
|
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||||
|
// cuts power. A device without a volatile cache reports success anyway.
|
||||||
|
const cdb = scsi.synchronizeCache10();
|
||||||
|
const ok = transact(&cdb, false, 0, 0);
|
||||||
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||||
|
},
|
||||||
else => return 0,
|
else => return 0,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -106,6 +106,12 @@ pub fn serialWrite(bytes: []const u8) void {
|
|||||||
serial.write(bytes);
|
serial.write(bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether a working UART was detected (loopback probe). When false the serial
|
||||||
|
/// sink is silently inert — a dead legacy COM1 costs nothing per byte.
|
||||||
|
pub fn serialPresent() bool {
|
||||||
|
return serial.present();
|
||||||
|
}
|
||||||
|
|
||||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||||
/// displays. The last-resort progress signal when there's no text output at all.
|
/// displays. The last-resort progress signal when there's no text output at all.
|
||||||
|
|||||||
@@ -7,8 +7,13 @@
|
|||||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||||
//! which the kernel heap will build on.
|
//! which the kernel heap will build on.
|
||||||
//!
|
//!
|
||||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
//! The physmap (the permanent window onto all physical RAM) is built with 2 MiB
|
||||||
//! negligible against available RAM.
|
//! huge pages wherever the range is 2 MiB-aligned, falling back to 4 KiB for the
|
||||||
|
//! unaligned edges. On a big machine that is the difference between ~16.7M page-
|
||||||
|
//! table entries (128 MiB of tables) and ~32K — it makes both the build and the
|
||||||
|
//! footprint scale sanely with RAM. Everything else (kernel segments, heap, user
|
||||||
|
//! space, on-demand MMIO) stays 4 KiB: precise, and the table memory is
|
||||||
|
//! negligible there.
|
||||||
|
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
@@ -22,10 +27,22 @@ const writable: u64 = 1 << 1;
|
|||||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||||
const pwt: u64 = 1 << 3; // page write-through
|
const pwt: u64 = 1 << 3; // page write-through
|
||||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||||
|
const page_size_bit: u64 = 1 << 7; // PS: this PDPT/PD entry is a 1 GiB/2 MiB leaf, not a pointer to the next table
|
||||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||||
const no_execute: u64 = 1 << 63;
|
const no_execute: u64 = 1 << 63;
|
||||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
|
// The PAT-index bit. In a 4 KiB PTE it is bit 7; in a huge leaf (2 MiB PDE / 1 GiB
|
||||||
|
// PDPTE) bit 7 is PS, so the PAT bit moves to bit 12. With PCD=PWT=0 this selects
|
||||||
|
// PAT entry 4, which `setupPat` programs to write-combining (see mapRangePhysmap).
|
||||||
|
const pte_pat: u64 = 1 << 7;
|
||||||
|
const huge_pat: u64 = 1 << 12;
|
||||||
|
const ia32_pat: u32 = 0x277;
|
||||||
|
|
||||||
|
/// The physmap's page size for 2 MiB-aligned RAM: one PD leaf covers this instead
|
||||||
|
/// of 512 PT entries. 4 KiB pages fill the unaligned edges (see mapRangePhysmap).
|
||||||
|
const huge_page_size: u64 = 2 << 20; // 2 MiB
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
// ELF segment flags (p_flags).
|
||||||
const pf_x: u32 = 1;
|
const pf_x: u32 = 1;
|
||||||
const pf_w: u32 = 2;
|
const pf_w: u32 = 2;
|
||||||
@@ -74,7 +91,15 @@ fn allocTable() u64 {
|
|||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||||
/// writable only if every level is; non-executable if any level is).
|
/// writable only if every level is; non-executable if any level is).
|
||||||
fn descend(entry: *u64) u64 {
|
fn descend(entry: *u64) u64 {
|
||||||
if (entry.* & present != 0) return entry.* & address_mask;
|
if (entry.* & present != 0) {
|
||||||
|
// A present-but-huge entry is a leaf, not a table: descending would read
|
||||||
|
// its 2 MiB/1 GiB data frame as a page table and corrupt RAM. This only
|
||||||
|
// fires on a bug — a 4 KiB map landing inside a physmap huge page — and a
|
||||||
|
// loud panic beats silent corruption. (The physmap and the 4 KiB regions
|
||||||
|
// live in disjoint PML4 slots, so it should never happen.)
|
||||||
|
if (entry.* & page_size_bit != 0) @panic("paging: descend through a huge-page leaf");
|
||||||
|
return entry.* & address_mask;
|
||||||
|
}
|
||||||
const frame = allocTable();
|
const frame = allocTable();
|
||||||
entry.* = frame | present | writable;
|
entry.* = frame | present | writable;
|
||||||
return frame;
|
return frame;
|
||||||
@@ -96,15 +121,39 @@ fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
|||||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map one 2 MiB huge page `virtual` -> `physical` with `flags` — a leaf at the PD
|
||||||
|
/// level (PS bit set), with no PT beneath it. Both addresses must be 2 MiB-aligned.
|
||||||
|
/// One of these replaces 512 `mapPage`s (and the PT frame they'd need).
|
||||||
|
fn mapHugePage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||||
|
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||||
|
@panic("paging: new higher-half PML4 entry after init");
|
||||||
|
const pdpt = descend(pml4e);
|
||||||
|
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = descend(pdpte);
|
||||||
|
tableAt(pd)[(virtual >> 21) & 0x1FF] = (physical & address_mask) | flags | present | page_size_bit;
|
||||||
|
}
|
||||||
|
|
||||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||||
/// window onto physical memory once the low identity map goes away.
|
/// window onto physical memory once the low identity map goes away. The 2 MiB-
|
||||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
/// aligned interior is mapped with huge pages; the unaligned head/tail with 4 KiB.
|
||||||
|
/// `write_combining` selects the WC memory type (setupPat's PAT entry 4) via the
|
||||||
|
/// PAT bit — bit 7 in a 4 KiB PTE, bit 12 in a huge leaf — for the framebuffer.
|
||||||
|
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64, write_combining: bool) void {
|
||||||
|
const pte_flags = if (write_combining) flags | pte_pat else flags;
|
||||||
|
const huge_flags = if (write_combining) flags | huge_pat else flags;
|
||||||
var address = physical_base & ~@as(u64, page_size - 1);
|
var address = physical_base & ~@as(u64, page_size - 1);
|
||||||
const end = physical_base + len;
|
const end = physical_base + len;
|
||||||
while (address < end) : (address += page_size) {
|
// Head: 4 KiB pages up to the next 2 MiB boundary.
|
||||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
while (address < end and address & (huge_page_size - 1) != 0) : (address += page_size)
|
||||||
}
|
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||||
|
// Interior: 2 MiB huge pages while a whole one still fits.
|
||||||
|
while (address + huge_page_size <= end) : (address += huge_page_size)
|
||||||
|
mapHugePage(pml4, boot_handoff.physicalToVirtual(address), address, huge_flags);
|
||||||
|
// Tail: 4 KiB pages for whatever is left.
|
||||||
|
while (address < end) : (address += page_size)
|
||||||
|
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||||
@@ -118,11 +167,26 @@ fn enableNx() void {
|
|||||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Program this core's PAT so entry 4 (selected by the PAT bit with PCD=PWT=0) is
|
||||||
|
/// **write-combining**, leaving the other seven at their reset types. Nothing else
|
||||||
|
/// in danos sets the PAT bit, so this changes no existing mapping — it only gives
|
||||||
|
/// the framebuffer a write-combining type, which turns its full-screen clear from
|
||||||
|
/// glacial (uncached writes to a GPU BAR, the real-hardware default via MTRRs) into
|
||||||
|
/// a batched burst. Must run on **every** core (PAT is per-logical-processor) — the
|
||||||
|
/// framebuffer mapping lives in the shared kernel half, so a core with the reset
|
||||||
|
/// PAT would see it as write-back and alias. Called from `init` (BSP) and each AP.
|
||||||
|
pub fn setupPat() void {
|
||||||
|
// Reset PAT is PA0=WB PA1=WT PA2=UC- PA3=UC PA4=WB PA5=WT PA6=UC- PA7=UC; flip
|
||||||
|
// PA4 from WB (0x06) to WC (0x01). Type codes: UC=0 WC=1 WT=4 WP=5 WB=6 UC-=7.
|
||||||
|
io.wrmsr(ia32_pat, 0x0007_0401_0007_0406);
|
||||||
|
}
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
/// Build the address space and switch onto it.
|
||||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||||
alloc_frame = allocFrame;
|
alloc_frame = allocFrame;
|
||||||
free_frame = freeFrame;
|
free_frame = freeFrame;
|
||||||
enableNx();
|
enableNx();
|
||||||
|
setupPat(); // BSP: PAT entry 4 = write-combining, for the framebuffer window
|
||||||
const pml4 = allocTable();
|
const pml4 = allocTable();
|
||||||
|
|
||||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||||
@@ -130,13 +194,15 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
|||||||
// mapped on demand (mapMmio) or explicitly below.
|
// mapped on demand (mapMmio) or explicitly below.
|
||||||
for (regions(boot_information.memory_map)) |r| {
|
for (regions(boot_information.memory_map)) |r| {
|
||||||
if (r.kind == .mmio) continue;
|
if (r.kind == .mmio) continue;
|
||||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||||
// the kernel touches directly), RW + NX.
|
// the kernel touches directly), RW + NX. The framebuffer is **write-
|
||||||
|
// combining** (see setupPat) so the console's full-screen clear is a burst,
|
||||||
|
// not millions of uncached single-word writes.
|
||||||
const fb = boot_information.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute, true);
|
||||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||||
|
|
||||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||||
@@ -338,8 +404,8 @@ fn freeSubtree(physical: u64, level: u32) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
/// clear. Walks the 4-level tables, stopping at a 2 MiB huge-page leaf (the physmap
|
||||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
/// uses them). Returns false if unmapped. Used for W^X checks in tests.
|
||||||
pub fn isExecutable(virtual: u64) bool {
|
pub fn isExecutable(virtual: u64) bool {
|
||||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return false;
|
if (pml4e & present == 0) return false;
|
||||||
@@ -347,6 +413,7 @@ pub fn isExecutable(virtual: u64) bool {
|
|||||||
if (pdpte & present == 0) return false;
|
if (pdpte & present == 0) return false;
|
||||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return false;
|
if (pde & present == 0) return false;
|
||||||
|
if (pde & page_size_bit != 0) return pde & no_execute == 0; // 2 MiB huge leaf
|
||||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return false;
|
if (pte & present == 0) return false;
|
||||||
return pte & no_execute == 0;
|
return pte & no_execute == 0;
|
||||||
@@ -385,9 +452,9 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
|||||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||||
/// needs the frame behind a user vaddr to free it).
|
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
||||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return null;
|
if (pml4e & present == 0) return null;
|
||||||
@@ -395,6 +462,8 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
|||||||
if (pdpte & present == 0) return null;
|
if (pdpte & present == 0) return null;
|
||||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return null;
|
if (pde & present == 0) return null;
|
||||||
|
if (pde & page_size_bit != 0) // 2 MiB huge leaf: frame base is bits 51:21
|
||||||
|
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return null;
|
if (pte & present == 0) return null;
|
||||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
|
|||||||
@@ -17,6 +17,12 @@ const Access = enum { port, mmio };
|
|||||||
var access: Access = .port;
|
var access: Access = .port;
|
||||||
var base: u64 = 0x3F8; // COM1
|
var base: u64 = 0x3F8; // COM1
|
||||||
|
|
||||||
|
/// Whether `init`/`reconfigure` found a *working* UART at `base`. False on a
|
||||||
|
/// legacy-free machine whose COM1 is decoded but dead: writing to it is then a
|
||||||
|
/// no-op, so `write` never spins waiting for a transmit register that will never
|
||||||
|
/// drain. Cleared until proven by the loopback probe.
|
||||||
|
var uart_present: bool = false;
|
||||||
|
|
||||||
fn portOut(p: u16, value: u8) void {
|
fn portOut(p: u16, value: u8) void {
|
||||||
asm volatile ("outb %[value], %[p]"
|
asm volatile ("outb %[value], %[p]"
|
||||||
:
|
:
|
||||||
@@ -57,6 +63,34 @@ pub fn init() void {
|
|||||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||||
setRegister(4, 0x0B); // RTS/DSR set
|
setRegister(4, 0x0B); // RTS/DSR set
|
||||||
|
uart_present = probe();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect a *working* UART by internal loopback: route the transmitter back to
|
||||||
|
/// the receiver (MCR bit 4), send a byte, and check it comes back. A port that is
|
||||||
|
/// merely decoded but has nothing behind it (the common case on a legacy-free
|
||||||
|
/// board that still answers I/O at 0x3F8) never echoes, so this returns false.
|
||||||
|
///
|
||||||
|
/// This matters for speed, not just correctness: a dead UART's line-status
|
||||||
|
/// register reads back 0x00, so its transmit-holding-empty bit never sets, and
|
||||||
|
/// `writeByte` would otherwise spin its full guard — tens of milliseconds — on
|
||||||
|
/// *every* logged byte. On real hardware that alone can add ~a minute to boot.
|
||||||
|
fn probe() bool {
|
||||||
|
const saved_mcr = register(4);
|
||||||
|
setRegister(4, 0x1E); // MCR: LOOP | OUT2 | OUT1 | RTS — internal loopback
|
||||||
|
setRegister(0, 0xAE); // push a distinctive byte into the loopback path
|
||||||
|
var guard: u32 = 0;
|
||||||
|
while (register(5) & 0x01 == 0 and guard < 10_000) : (guard += 1) {} // await Data Ready
|
||||||
|
const echo = register(0);
|
||||||
|
setRegister(4, saved_mcr); // restore the modem-control lines
|
||||||
|
return echo == 0xAE;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a working UART was detected (see `probe`). The log sink stays
|
||||||
|
/// registered regardless — it simply does nothing until this is true — so a UART
|
||||||
|
/// that only `reconfigure` discovers (via SPCR) still starts logging.
|
||||||
|
pub fn present() bool {
|
||||||
|
return uart_present;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||||
@@ -70,15 +104,19 @@ pub fn reconfigure(is_mmio: bool, address: u64) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn writeByte(c: u8) void {
|
fn writeByte(c: u8) void {
|
||||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
// Wait for the transmit-holding register to empty. `write` only reaches here
|
||||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
// for a UART the loopback probe proved live, so this bounds a momentary stall
|
||||||
|
// (e.g. deasserted flow control), not an absent port: ~5000 legacy-port reads
|
||||||
|
// is a few ms — comfortably longer than one 38400-baud byte-time (~260 µs).
|
||||||
var guard: u32 = 0;
|
var guard: u32 = 0;
|
||||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
while (register(5) & 0x20 == 0 and guard < 5_000) : (guard += 1) {}
|
||||||
setRegister(0, c);
|
setRegister(0, c);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
/// Write bytes, translating LF to CRLF so terminals and logs line up. A no-op
|
||||||
|
/// when no working UART was detected, so a dead COM1 costs nothing per byte.
|
||||||
pub fn write(bytes: []const u8) void {
|
pub fn write(bytes: []const u8) void {
|
||||||
|
if (!uart_present) return;
|
||||||
for (bytes) |c| {
|
for (bytes) |c| {
|
||||||
if (c == '\n') writeByte('\r');
|
if (c == '\n') writeByte('\r');
|
||||||
writeByte(c);
|
writeByte(c);
|
||||||
|
|||||||
@@ -173,6 +173,7 @@ fn delayMicros(us: u64) void {
|
|||||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||||
const cpu = boot_index;
|
const cpu = boot_index;
|
||||||
|
paging.setupPat(); // this core's PAT: entry 4 = write-combining, to match the BSP
|
||||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||||
idt.loadOnThisCpu(); // the shared IDT
|
idt.loadOnThisCpu(); // the shared IDT
|
||||||
|
|||||||
+31
-16
@@ -60,8 +60,16 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||||
// with port-0x80 checkpoints as the only progress signal.
|
// with port-0x80 checkpoints as the only progress signal.
|
||||||
architecture.serialInit();
|
//
|
||||||
log.addSink(architecture.serialWrite);
|
// Serial is compiled in only under -Dserial (build.zig): a real machine often
|
||||||
|
// has no live legacy COM1, and the log survives in the RAM buffer (below) and
|
||||||
|
// is flushed to disk — so serial is now a QEMU/dev convenience the flashable
|
||||||
|
// image leaves out. When it *is* built in, `serialInit`'s loopback probe still
|
||||||
|
// guards against a dead port (so a -Dserial image is safe on real hardware).
|
||||||
|
if (build_options.serial) {
|
||||||
|
architecture.serialInit();
|
||||||
|
log.addSink(architecture.serialWrite);
|
||||||
|
}
|
||||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||||
// Retain the whole stream in a RAM buffer too, so a user program can later
|
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||||
// read it back (klog_read) and persist the boot log to disk — the only way to
|
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||||
@@ -72,8 +80,14 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||||
|
//
|
||||||
|
// The console is brought up *after* paging (below), not here: its one-time
|
||||||
|
// full-screen clear then runs on the kernel's **write-combining** mapping of the
|
||||||
|
// framebuffer instead of the loader's uncached one — a fast burst rather than
|
||||||
|
// millions of uncached writes on real hardware. Until then, on-screen output is
|
||||||
|
// absent (an early panic still lands in the serial/RAM log); the trade is worth
|
||||||
|
// a near-instant boot. `console.write` is a safe no-op while the console is down.
|
||||||
const fb = boot_information.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
console.init(fb);
|
|
||||||
|
|
||||||
log.checkpoint(cp_entry);
|
log.checkpoint(cp_entry);
|
||||||
|
|
||||||
@@ -83,10 +97,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
architecture.init();
|
architecture.init();
|
||||||
|
|
||||||
status("/system/kernel: initialising kernel...\n");
|
status("/system/kernel: initialising kernel...\n");
|
||||||
log.write(if (console.present())
|
if (build_options.serial) log.write(if (architecture.serialPresent())
|
||||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
"/system/kernel: serial console online (COM1)\n"
|
||||||
else
|
else
|
||||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
"/system/kernel: no serial UART (COM1 absent) -> log kept in RAM/debugcon\n");
|
||||||
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||||
@@ -142,6 +156,15 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||||
|
|
||||||
|
// Now on our own tables, the framebuffer window is write-combining: bring up
|
||||||
|
// the on-screen console and clear it (a fast burst here, not the loader's
|
||||||
|
// uncached crawl). From here `status` reaches the screen as well as the log.
|
||||||
|
console.init(fb);
|
||||||
|
log.write(if (console.present())
|
||||||
|
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||||
|
else
|
||||||
|
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||||
heap.init();
|
heap.init();
|
||||||
log.checkpoint(cp_heap);
|
log.checkpoint(cp_heap);
|
||||||
@@ -176,21 +199,13 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||||
irq.init();
|
irq.init();
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||||
|
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||||
const pw = platform.powerInformation();
|
const pw = platform.powerInformation();
|
||||||
log.write("/system/kernel: power\n");
|
log.write("/system/kernel: power\n");
|
||||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||||
if (pw.s5) |s| {
|
|
||||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
|
||||||
} else {
|
|
||||||
log.write(" S5 slp_typ : (not found)\n");
|
|
||||||
}
|
|
||||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||||
|
|
||||||
// AML namespace parse integrity: consumed should equal total.
|
|
||||||
const am = platform.amlStats();
|
|
||||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
|
||||||
|
|
||||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||||
|
|||||||
+23
-34
@@ -181,10 +181,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
containmentTest();
|
containmentTest();
|
||||||
} else if (eql(case, "device-manager")) {
|
} else if (eql(case, "device-manager")) {
|
||||||
deviceManagerTest(boot_information);
|
deviceManagerTest(boot_information);
|
||||||
} else if (eql(case, "poweroff")) {
|
|
||||||
powerTest(.off);
|
|
||||||
} else if (eql(case, "reboot")) {
|
} else if (eql(case, "reboot")) {
|
||||||
powerTest(.reboot);
|
rebootTest();
|
||||||
} else {
|
} else {
|
||||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||||
}
|
}
|
||||||
@@ -201,15 +199,14 @@ fn platformHal() platform.Hal {
|
|||||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
||||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
||||||
/// transition failed and we emit a FAIL result.
|
/// transition failed and we emit a FAIL result.
|
||||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
// Soft-off (S5) is no longer a kernel operation — the ring-3 acpi service owns it
|
||||||
const name = if (action == .off) "poweroff" else "reboot";
|
// (exercised end-to-end by `orderly-shutdown`). Reboot stays in the kernel (FADT
|
||||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
// reset register, no AML), so it keeps its own case.
|
||||||
|
fn rebootTest() void {
|
||||||
|
log("DANOS-TEST-BEGIN: reboot\n", .{});
|
||||||
const hal = platformHal();
|
const hal = platformHal();
|
||||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
log("DANOS-POWER: attempting reboot\n", .{});
|
||||||
switch (action) {
|
platform.reboot(hal);
|
||||||
.off => platform.shutdown(hal),
|
|
||||||
.reboot => platform.reboot(hal),
|
|
||||||
}
|
|
||||||
check("power transition took effect", false);
|
check("power transition took effect", false);
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -276,22 +273,19 @@ fn timer() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Verify device discovery populated the platform facts the rest of the kernel
|
/// Verify device discovery populated the platform facts the rest of the kernel
|
||||||
/// depends on — the results ACPI parsing stashed in globals at boot. These are
|
/// depends on — the results the static ACPI tables stashed in globals at boot.
|
||||||
/// stable for the QEMU q35 + OVMF machine the harness runs, and span the tables:
|
/// These are stable for the QEMU q35 + OVMF machine the harness runs, and span the
|
||||||
/// MADT (LAPIC base, CPU count), FADT (PM/reset registers), and the AML parse
|
/// tables: MADT (LAPIC base, CPU count) and FADT (PM/reset registers). The kernel
|
||||||
/// (the sleep type, plus the integrity check that every byte was consumed).
|
/// no longer interprets AML — sleep types are the ring-3 acpi service's concern.
|
||||||
fn discoveryTest() void {
|
fn discoveryTest() void {
|
||||||
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
||||||
const pinfo = platform.platformInformation();
|
const pinfo = platform.platformInformation();
|
||||||
const pw = platform.powerInformation();
|
const pw = platform.powerInformation();
|
||||||
const am = platform.amlStats();
|
|
||||||
|
|
||||||
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
||||||
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
||||||
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
||||||
check("reset register supported (FADT)", pw.reset_supported);
|
check("reset register supported (FADT)", pw.reset_supported);
|
||||||
check("S5 sleep type found (AML)", pw.s5 != null);
|
|
||||||
check("AML parsed completely (consumed == total)", am.total > 0 and am.consumed == am.total);
|
|
||||||
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
||||||
|
|
||||||
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
||||||
@@ -2095,11 +2089,11 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// M20.1: the ring-3 AML parse agrees with the kernel's. The manager spawns
|
/// M20.1: the ring-3 AML parse works. The manager spawns the discovery service
|
||||||
/// the discovery service (the acpi build variant); it claims the acpi-tables
|
/// (the acpi build variant); it claims the acpi-tables node, maps the blobs,
|
||||||
/// node, maps the blobs, parses them, and logs its Device count — which must
|
/// parses them, and self-verifies it found at least a floor of Device objects,
|
||||||
/// equal what the kernel's own parse produced (the equivalence that licenses
|
/// printing "acpi-parse: ok". The kernel no longer parses AML, so there is no
|
||||||
/// retiring the kernel's device build in M20.3).
|
/// kernel count to compare against — the ring-3 parse is now the only one.
|
||||||
fn acpiParseTest(boot_information: *const BootInformation) void {
|
fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
@@ -2114,23 +2108,18 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
// The kernel's own count, from the namespace it already built for \_S5.
|
// Spawn the discovery service directly with a device-count *floor* as argv:
|
||||||
const kernel_devices = platform.amlDeviceCount();
|
// it parses the blobs in ring 3 and self-verifies it found at least that many
|
||||||
check("the kernel namespace has devices to compare against", kernel_devices >= 1);
|
// Device objects, printing "acpi-parse: ok". A floor of 1 just proves the
|
||||||
|
// parser ran and produced a namespace (the QEMU q35 DSDT has dozens). The
|
||||||
// Spawn the discovery service directly with that count as argv: it parses
|
// marker is deterministic — no racing the shared serial buffer.
|
||||||
// the same blobs in ring 3 and self-verifies, printing "acpi-parse: ok" iff
|
|
||||||
// the counts match. The harness's expect regex is that marker — deterministic,
|
|
||||||
// no racing the shared serial buffer.
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
var count_text: [16]u8 = undefined;
|
|
||||||
const count_arg = std.fmt.bufPrint(&count_text, "{d}", .{kernel_devices}) catch "0";
|
|
||||||
var spawned = false;
|
var spawned = false;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(item.name, "discovery")) continue;
|
if (!eql(item.name, "discovery")) continue;
|
||||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", count_arg }, scheduler.currentId(), null) catch 0;
|
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
|
||||||
spawned = true;
|
spawned = true;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -103,9 +103,11 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||||
// own device count to self-verify against — deterministic, no log-scraping.
|
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||||
|
// devices is the check. Deterministic, no log-scraping.
|
||||||
|
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||||
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||||
@@ -162,11 +164,11 @@ pub fn main(init: runtime.process.Init) void {
|
|||||||
var namespace = result.namespace;
|
var namespace = result.namespace;
|
||||||
const devices = aml.deviceCount(&namespace);
|
const devices = aml.deviceCount(&namespace);
|
||||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||||
if (expected) |want| {
|
if (floor) |minimum| {
|
||||||
if (devices == want) {
|
if (devices >= minimum) {
|
||||||
_ = runtime.system.write("acpi-parse: ok\n");
|
_ = runtime.system.write("acpi-parse: ok\n");
|
||||||
} else {
|
} else {
|
||||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
writeLine("acpi-parse: too few (ring-3 {d} < floor {d})\n", .{ devices, minimum });
|
||||||
}
|
}
|
||||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||||
while (true) runtime.system.sleep(1000);
|
while (true) runtime.system.sleep(1000);
|
||||||
|
|||||||
@@ -16,6 +16,10 @@ pub const Operation = enum(u32) {
|
|||||||
read = 1,
|
read = 1,
|
||||||
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||||
write = 2,
|
write = 2,
|
||||||
|
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||||
|
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||||
|
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||||
|
flush = 3,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub const Request = extern struct {
|
pub const Request = extern struct {
|
||||||
|
|||||||
@@ -40,11 +40,16 @@ const IpcBlock = struct {
|
|||||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||||
@memcpy(destination[0..512], buffer[0..512]);
|
@memcpy(destination[0..512], buffer[0..512]);
|
||||||
return self.device.write(lba, 1, self.bounce.physical);
|
if (!self.device.write(lba, 1, self.bounce.physical)) return false;
|
||||||
|
device_dirty = true; // a block reached the device; a close will flush it
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
var ipc_block: IpcBlock = undefined;
|
var ipc_block: IpcBlock = undefined;
|
||||||
|
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||||
|
// file close — so writes are committed to stable media before a power-off.
|
||||||
|
var device_dirty: bool = false;
|
||||||
var filesystem: engine.FileSystem = undefined;
|
var filesystem: engine.FileSystem = undefined;
|
||||||
|
|
||||||
// Open handles the VFS holds against this backend: each maps a node id to a
|
// Open handles the VFS holds against this backend: each maps a node id to a
|
||||||
@@ -192,6 +197,14 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
|||||||
},
|
},
|
||||||
.close => {
|
.close => {
|
||||||
if (openAt(request.node)) |o| o.used = false;
|
if (openAt(request.node)) |o| o.used = false;
|
||||||
|
// Durable-on-close: if any block reached the device since the last
|
||||||
|
// flush, commit its cache to stable media now (best-effort). This is
|
||||||
|
// what makes init's shutdown log flush survive a real power-off, and is
|
||||||
|
// the right default for removable media the user may unplug.
|
||||||
|
if (device_dirty) {
|
||||||
|
_ = ipc_block.device.flush();
|
||||||
|
device_dirty = false;
|
||||||
|
}
|
||||||
return writeReply(out, .{ .status = 0 }, &.{});
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
},
|
},
|
||||||
.mkdir => {
|
.mkdir => {
|
||||||
|
|||||||
+7
-5
@@ -365,7 +365,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
"expect": r"acpi-parse: ok",
|
"expect": r"acpi-parse: ok",
|
||||||
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
"fail": r"acpi-parse: too few|DANOS-TEST-RESULT: FAIL"},
|
||||||
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||||
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||||
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
||||||
@@ -492,9 +492,8 @@ CASES = [
|
|||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
||||||
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
||||||
{"name": "poweroff",
|
# Soft-off (S5) is owned by the ring-3 acpi service now (see orderly-shutdown);
|
||||||
"expect": r"DANOS-POWER: attempting poweroff",
|
# the kernel keeps only reboot (FADT reset register, no AML).
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
|
||||||
{"name": "reboot",
|
{"name": "reboot",
|
||||||
"expect": r"DANOS-POWER: attempting reboot",
|
"expect": r"DANOS-POWER: attempting reboot",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
@@ -504,7 +503,10 @@ TIMEOUT = 30 # seconds per case
|
|||||||
|
|
||||||
|
|
||||||
def build(arch, case):
|
def build(arch, case):
|
||||||
cmd = ["zig", "build", f"-Dtest-case={case}"] + arch["zig_flags"]
|
# -Dserial: the harness asserts on markers the kernel writes to serial0, so the
|
||||||
|
# serial log sink must be compiled in. It is off by default (a flashed real-
|
||||||
|
# hardware image keeps its log in RAM instead; see build.zig / serial.zig).
|
||||||
|
cmd = ["zig", "build", f"-Dtest-case={case}", "-Dserial=true"] + arch["zig_flags"]
|
||||||
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
|
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
|
||||||
if r.returncode != 0:
|
if r.returncode != 0:
|
||||||
return r.stderr.strip() or r.stdout.strip()
|
return r.stderr.strip() or r.stdout.strip()
|
||||||
|
|||||||
Reference in New Issue
Block a user