Compare commits
25
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
24c49f56e1 | ||
|
|
0217662808 | ||
|
|
4231301896 | ||
|
|
58927ed7e5 | ||
|
|
6e0e0a62c6 | ||
|
|
88ad432758 | ||
|
|
9333d0572f | ||
|
|
f3342118f5 | ||
|
|
2723b6f778 | ||
|
|
a01a4f3b3d | ||
|
|
e1605e3235 | ||
|
|
1e80c57484 | ||
|
|
c3e9c59086 | ||
|
|
88ed3c5417 | ||
|
|
3fc2d5b083 | ||
|
|
105203b447 | ||
|
|
f9cf0007c5 | ||
|
|
69b018cc32 | ||
|
|
cd812cc00e | ||
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f |
+12
-3
@@ -2,6 +2,7 @@ const std = @import("std");
|
|||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const build_options = @import("build_options");
|
||||||
const BootInformation = boot_handoff.BootInformation;
|
const BootInformation = boot_handoff.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
@@ -84,7 +85,7 @@ fn boot() !noreturn {
|
|||||||
// the map and exiting would invalidate the map key.
|
// the map and exiting would invalidate the map key.
|
||||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
log("EFI: kernel loaded, exiting boot services\r\n");
|
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||||
boot_information.memory_map = try exitBootServices(bs);
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
@@ -395,7 +396,7 @@ fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !
|
|||||||
const image = try loadFile(bs, init_file_name);
|
const image = try loadFile(bs, init_file_name);
|
||||||
boot_information.init_base = @intFromPtr(image.ptr);
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
boot_information.init_len = image.len;
|
boot_information.init_len = image.len;
|
||||||
log("EFI: /system/services/init loaded\r\n");
|
progress("EFI: /system/services/init loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||||
@@ -403,7 +404,7 @@ fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInfo
|
|||||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||||
boot_information.initial_ramdisk_len = image.len;
|
boot_information.initial_ramdisk_len = image.len;
|
||||||
log("EFI: initial_ramdisk loaded\r\n");
|
progress("EFI: initial_ramdisk loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
@@ -561,6 +562,14 @@ fn log(comptime message: []const u8) void {
|
|||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||||
|
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||||
|
/// `log` directly and always show, so a failed boot still explains itself.
|
||||||
|
fn progress(comptime message: []const u8) void {
|
||||||
|
if (!build_options.serial) return;
|
||||||
|
log(message);
|
||||||
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
|
|||||||
@@ -96,6 +96,105 @@ fn addUserBinary(
|
|||||||
return exe;
|
return exe;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||||
|
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||||
|
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||||
|
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||||
|
const KernelModules = struct {
|
||||||
|
boot_handoff: *std.Build.Module,
|
||||||
|
abi: *std.Build.Module,
|
||||||
|
device_abi: *std.Build.Module,
|
||||||
|
architecture: *std.Build.Module,
|
||||||
|
platform: *std.Build.Module,
|
||||||
|
parameters: *std.Build.Module,
|
||||||
|
initial_ramdisk: *std.Build.Module,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||||
|
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||||
|
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||||
|
/// build option baked into `build_options`.
|
||||||
|
fn addKernel(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_target: std.Build.ResolvedTarget,
|
||||||
|
optimize: std.builtin.OptimizeMode,
|
||||||
|
modules: KernelModules,
|
||||||
|
test_case: ?[]const u8,
|
||||||
|
serial: bool,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||||
|
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||||
|
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||||
|
const build_options = b.addOptions();
|
||||||
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
|
build_options.addOption(bool, "serial", serial);
|
||||||
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = "kernel",
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||||
|
.target = kernel_target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
|
.stack_protector = false,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||||
|
.{ .name = "abi", .module = modules.abi },
|
||||||
|
.{ .name = "device-abi", .module = modules.device_abi },
|
||||||
|
.{ .name = "architecture", .module = modules.architecture },
|
||||||
|
.{ .name = "platform", .module = modules.platform },
|
||||||
|
.{ .name = "parameters", .module = modules.parameters },
|
||||||
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
|
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||||
|
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||||
|
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||||
|
/// bundle its own kernel while sharing the (serial-independent) loader, init, and
|
||||||
|
/// ramdisk. Returns the image's LazyPath.
|
||||||
|
fn addBootImage(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_bin: std.Build.LazyPath,
|
||||||
|
efi_bin: std.Build.LazyPath,
|
||||||
|
init_bin: std.Build.LazyPath,
|
||||||
|
initial_ramdisk_img: std.Build.LazyPath,
|
||||||
|
) std.Build.LazyPath {
|
||||||
|
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||||
|
mk_fat.addArg("64"); // MiB
|
||||||
|
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||||
|
mk_fat.addFileArg(efi_bin);
|
||||||
|
mk_fat.addArg("system/kernel");
|
||||||
|
mk_fat.addFileArg(kernel_bin);
|
||||||
|
mk_fat.addArg("system/services/init");
|
||||||
|
mk_fat.addFileArg(init_bin);
|
||||||
|
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||||
|
mk_fat.addFileArg(initial_ramdisk_img);
|
||||||
|
return fat_image;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
@@ -249,6 +348,20 @@ pub fn build(b: *std.Build) void {
|
|||||||
// The block protocol, so runtime.block (the block-device client) can speak it.
|
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||||
runtime_module.addImport("block-protocol", block_protocol_module);
|
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||||
|
|
||||||
|
// The display protocol, so runtime.display (the compositor client) and the display
|
||||||
|
// service both speak it through the runtime, like the other protocol modules.
|
||||||
|
const display_protocol_module = b.addModule("display-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/display/protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||||
|
|
||||||
|
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||||
|
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||||
|
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||||
|
|
||||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||||
const power_protocol_module = b.addModule("power-protocol", .{
|
const power_protocol_module = b.addModule("power-protocol", .{
|
||||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||||
@@ -285,9 +398,12 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
// The serial-console log sink. Off by default: a real machine often has no
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||||
const build_options_module = build_options.createModule();
|
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||||
|
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||||
|
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||||
|
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -299,41 +415,17 @@ pub fn build(b: *std.Build) void {
|
|||||||
.abi = .none,
|
.abi = .none,
|
||||||
});
|
});
|
||||||
|
|
||||||
const exe = b.addExecutable(.{
|
const kernel_modules = KernelModules{
|
||||||
.name = "kernel",
|
.boot_handoff = boot_handoff_module,
|
||||||
.root_module = b.createModule(.{
|
.abi = abi_module,
|
||||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
.device_abi = device_abi_module,
|
||||||
.target = kernel_target,
|
.architecture = architecture_module,
|
||||||
.optimize = optimize,
|
.platform = platform_module,
|
||||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
.parameters = parameters_module,
|
||||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
.initial_ramdisk = initial_ramdisk_module,
|
||||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
};
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||||
.stack_protector = false,
|
|
||||||
.imports = &.{
|
|
||||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
|
||||||
.{ .name = "abi", .module = abi_module },
|
|
||||||
.{ .name = "device-abi", .module = device_abi_module },
|
|
||||||
.{ .name = "architecture", .module = architecture_module },
|
|
||||||
.{ .name = "platform", .module = platform_module },
|
|
||||||
.{ .name = "parameters", .module = parameters_module },
|
|
||||||
.{ .name = "build_options", .module = build_options_module },
|
|
||||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
|
||||||
},
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
|
||||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
|
||||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
|
||||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
|
||||||
exe.use_llvm = true;
|
|
||||||
exe.use_lld = true;
|
|
||||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
|
||||||
// linker's AT() clauses give each segment a low physical load address
|
|
||||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
|
||||||
exe.image_base = 0xFFFFFFFF80100000;
|
|
||||||
|
|
||||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
@@ -381,6 +473,11 @@ pub fn build(b: *std.Build) void {
|
|||||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||||
|
const display_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||||
|
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||||
|
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||||
|
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||||
|
const shm_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-client", "system/services/shm-client/shm-client.zig");
|
||||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||||
// The PCI bus driver decodes each function's class triple to human names in its
|
// The PCI bus driver decodes each function's class triple to human names in its
|
||||||
@@ -448,6 +545,16 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(fat_exe.getEmittedBin());
|
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||||
mk_run.addArg("fat-test");
|
mk_run.addArg("fat-test");
|
||||||
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("display");
|
||||||
|
mk_run.addFileArg(display_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("display-demo");
|
||||||
|
mk_run.addFileArg(display_demo_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("virtio-gpu");
|
||||||
|
mk_run.addFileArg(virtio_gpu_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("shm-server");
|
||||||
|
mk_run.addFileArg(shm_server_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("shm-client");
|
||||||
|
mk_run.addFileArg(shm_client_exe.getEmittedBin());
|
||||||
mk_run.addArg("pci-bus");
|
mk_run.addArg("pci-bus");
|
||||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||||
mk_run.addArg("crash-test");
|
mk_run.addArg("crash-test");
|
||||||
@@ -485,6 +592,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ usb_hid_mouse_exe, "system/drivers" },
|
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||||
.{ usb_storage_exe, "system/drivers" },
|
.{ usb_storage_exe, "system/drivers" },
|
||||||
.{ fat_exe, "system/services" },
|
.{ fat_exe, "system/services" },
|
||||||
|
.{ display_exe, "system/services" },
|
||||||
.{ log_flush_exe, "system/services" },
|
.{ log_flush_exe, "system/services" },
|
||||||
}) |entry| {
|
}) |entry| {
|
||||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||||
@@ -498,6 +606,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||||
// Each is its own binary/entry (a loader is built for its own target); today
|
// Each is its own binary/entry (a loader is built for its own target); today
|
||||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||||
|
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||||
|
// which firmware may mirror to a serial console) are silenced by default — a
|
||||||
|
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||||
|
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||||
|
const loader_options = b.addOptions();
|
||||||
|
loader_options.addOption(bool, "serial", serial);
|
||||||
|
const loader_options_module = loader_options.createModule();
|
||||||
const efiexe = b.addExecutable(.{
|
const efiexe = b.addExecutable(.{
|
||||||
.name = "BOOTX64",
|
.name = "BOOTX64",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
@@ -510,6 +625,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.imports = &.{
|
.imports = &.{
|
||||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
|
.{ .name = "build_options", .module = loader_options_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
@@ -525,21 +641,17 @@ pub fn build(b: *std.Build) void {
|
|||||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
|
||||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
|
||||||
mk_fat.addArg("64"); // MiB
|
|
||||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
|
||||||
mk_fat.addFileArg(efiexe.getEmittedBin());
|
|
||||||
mk_fat.addArg("system/kernel");
|
|
||||||
mk_fat.addFileArg(exe.getEmittedBin());
|
|
||||||
mk_fat.addArg("system/services/init");
|
|
||||||
mk_fat.addFileArg(init_exe.getEmittedBin());
|
|
||||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
|
||||||
mk_fat.addFileArg(initial_ramdisk_img);
|
|
||||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|
||||||
|
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||||
|
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||||
|
// log captured to serial0 — without baking serial into the image users flash.
|
||||||
|
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||||
|
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||||
|
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||||
|
|
||||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||||
@@ -611,8 +723,9 @@ pub fn build(b: *std.Build) void {
|
|||||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||||
|
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||||
run_efi.addArg("-drive");
|
run_efi.addArg("-drive");
|
||||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image);
|
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||||
run_efi.addArgs(&.{
|
run_efi.addArgs(&.{
|
||||||
"-device",
|
"-device",
|
||||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||||
@@ -634,8 +747,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||||
run_efi.step.dependOn(b.getInstallStep());
|
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||||
|
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||||
|
// scratch dir first.
|
||||||
run_efi.step.dependOn(&make_log_dir.step);
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
@@ -674,6 +789,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||||
|
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||||
|
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||||
|
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||||
|
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||||
}) |root| {
|
}) |root| {
|
||||||
const mod_tests = b.addTest(.{
|
const mod_tests = b.addTest(.{
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
|
|||||||
+9
-1
@@ -69,7 +69,15 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||||
service layered on top.
|
service layered on top.
|
||||||
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
19. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||||
|
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||||
|
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||||
|
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||||
|
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||||
|
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||||
|
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||||
|
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md).
|
||||||
|
20. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
|
|||||||
@@ -0,0 +1,181 @@
|
|||||||
|
# Display service — build plan (v1: the dumb-framebuffer compositor)
|
||||||
|
|
||||||
|
The ordered, checkpointable build-out for [display.md](display.md). Each milestone is
|
||||||
|
small, lands on its own, and ends in a **verifiable gate** — shaped for a `/loop` run.
|
||||||
|
Read [display.md](display.md) first for the *why*; this is the *what* and the *order*.
|
||||||
|
|
||||||
|
## Locked decisions (do not relitigate)
|
||||||
|
|
||||||
|
- **Handoff = device node + write-combining `mmio_map`.** The kernel seeds a synthetic
|
||||||
|
`display0` node from `BootInformation.framebuffer`; the service claims + WC-maps it.
|
||||||
|
(Not a bespoke `framebuffer_map` syscall — the device route inherits ownership,
|
||||||
|
release-on-death, and re-claim-on-restart.)
|
||||||
|
- **v1 = the full compositor pipeline on the dumb framebuffer.** One `display` service
|
||||||
|
owns the LFB + a cacheable back buffer + a layer stack; double-buffer + damage-driven
|
||||||
|
present; clients draw via server-side commands. **No** runtime mode-setting, **no**
|
||||||
|
shared-memory surfaces — both deferred (see display.md, "What v1 does not do").
|
||||||
|
|
||||||
|
## Conventions
|
||||||
|
|
||||||
|
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||||
|
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||||
|
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||||
|
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||||
|
`runtime` module.
|
||||||
|
|
||||||
|
## How to verify along the way
|
||||||
|
|
||||||
|
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||||
|
pitch/format blits are all host-testable with a fake framebuffer).
|
||||||
|
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||||
|
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||||
|
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||||
|
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## D1 — The handoff primitive (kernel) ✅
|
||||||
|
|
||||||
|
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||||
|
|
||||||
|
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||||
|
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||||
|
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||||
|
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||||
|
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||||
|
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||||
|
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||||
|
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||||
|
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||||
|
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||||
|
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||||
|
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||||
|
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||||
|
|
||||||
|
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||||
|
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||||
|
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||||
|
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||||
|
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||||
|
screenshot-of-a-fill gate because it proves the *actual* WC property headlessly; the
|
||||||
|
visible fill folds into D2's gate (the service clears the screen through the back buffer).
|
||||||
|
Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `device-list`,
|
||||||
|
`device-manager` all still pass with the +1 device in the table.
|
||||||
|
|
||||||
|
## D2 — Service skeleton, protocol, runtime module ✅
|
||||||
|
|
||||||
|
Stand up the named service and the double-buffer, no layers yet.
|
||||||
|
|
||||||
|
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||||
|
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||||
|
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||||
|
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||||
|
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||||
|
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||||
|
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||||
|
until D3. Init clears the back buffer and presents it — the double-buffer path.
|
||||||
|
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||||
|
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||||
|
lookup with retry (model: block.zig).
|
||||||
|
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||||
|
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||||
|
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||||
|
`/system/services/display`.
|
||||||
|
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||||
|
a fixed kernel-stack `frames` array. Rewrote `systemMmap` to map page-by-page with
|
||||||
|
rollback (no scratch array) and raised the cap to 8192 pages (32 MiB) — enough for a
|
||||||
|
4K back buffer. A real limitation met, exactly the kind this project chases.
|
||||||
|
|
||||||
|
**Gate (met, automated):** `python3 test/qemu_test.py display-service` spawns the
|
||||||
|
compositor and matches its own serial heartbeats — `display: online {w}x{h} pitch …`
|
||||||
|
followed by `display: presented frame 0` — which it prints only after the whole
|
||||||
|
claim → WC-map → back-buffer → clear → present chain succeeds (matched on serial like the
|
||||||
|
fault cases, since a lone blocking service can't reschedule the in-kernel test context to
|
||||||
|
poll). Regression-checked: `usermem`, `heap` (the `mmap` rewrite), `init` (the boot-list
|
||||||
|
addition), and D1's `display` all still pass.
|
||||||
|
|
||||||
|
## D3 — Layer stack + compositor + damage present ✅
|
||||||
|
|
||||||
|
The heart: composite an ordered layer stack, present only what changed.
|
||||||
|
|
||||||
|
- [x] A layer table (16 slots): each `Layer` = position, z, visible, a server-owned
|
||||||
|
`mmap`'d surface (freed on `destroy_layer`). `damage` accumulates the dirty screen
|
||||||
|
region since the last present.
|
||||||
|
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||||
|
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||||
|
`damage`, `present`.
|
||||||
|
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||||
|
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||||
|
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||||
|
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||||
|
Colour packing (rgbx/bgrx) is `protocol.pack`, also host-tested.
|
||||||
|
- [x] Host tests (`zig build test`, green): rect intersect/unite, `fillRect` clipping +
|
||||||
|
`stride > width` padding, `composite` overlap-shows-top + damage clipping, `blitTile`
|
||||||
|
unaligned read + clipping, and `pack` for both pixel formats.
|
||||||
|
|
||||||
|
**Gate (met):** `zig build test` green for the compositor + pack unit tests, **and** the
|
||||||
|
`display-service` case's startup self-check composites two overlapping layers on the real
|
||||||
|
framebuffer and reads back the composited pixels — overlap = top layer, outside = bottom
|
||||||
|
layer — logging `display: compositor self-check ok` (matched by the harness).
|
||||||
|
|
||||||
|
## D4 — Client API + the demo client ✅
|
||||||
|
|
||||||
|
Prove the pipeline end-to-end from a separate process.
|
||||||
|
|
||||||
|
- [x] Finished [runtime/display.zig](../library/runtime/runtime.zig): a `Layer` handle with
|
||||||
|
`fill` / `blitTile` (inline tile) / `configure` (move/restack/show) / `damage` /
|
||||||
|
`destroy`, `createLayer`, and a `color(r,g,b)` helper (caches the mode, packs via
|
||||||
|
`protocol.pack`). Coordinates are signed over the wire (`@bitCast` both ways).
|
||||||
|
- [x] `system/services/display-demo/`: a hardware-free client (the `input-source` analog)
|
||||||
|
— a full-screen wallpaper layer, a rectangle that slides back and forth (moved by
|
||||||
|
`configure` each frame, so the compositor repaints old + new), and a cursor layer;
|
||||||
|
presents in a loop paced by `runtime.time`. Wired into build + initial-ramdisk.
|
||||||
|
- [x] **Bug this surfaced:** `protocol.message_maximum` was 4096, but the kernel caps
|
||||||
|
every IPC message at `MESSAGE_MAXIMUM` = 256 — so `replyWait` rejected the oversized
|
||||||
|
receive buffer with `-E2BIG` and the serve loop had been *spinning* since D2 (unseen,
|
||||||
|
as D2/D3 matched init-time heartbeats). Set it to 256; `blit_tile` is now explicitly
|
||||||
|
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shm surface.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py display-demo` spawns the service + `display-demo`;
|
||||||
|
the demo drives a run of frames of motion through the layer client API and logs
|
||||||
|
`display-demo: ok` (the visible motion is a screenshot via `zig build run-x86-64`).
|
||||||
|
Regression-checked: `zig build test`, `display` (D1), and `display-service` (D2/D3) all
|
||||||
|
still pass, and the default `zig build` is clean.
|
||||||
|
|
||||||
|
## D5 — Test cases + docs ✅
|
||||||
|
|
||||||
|
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||||
|
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||||
|
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||||
|
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||||
|
the pure host tests (`zig build test`).
|
||||||
|
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||||
|
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||||
|
memory marked DONE with the commits.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||||
|
`zig build test` is green, and the default `zig build` is clean.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## v1 status: complete
|
||||||
|
|
||||||
|
D1–D5 done. The display service is a working framebuffer compositor: it owns the
|
||||||
|
framebuffer (write-combining), composites a z-ordered layer stack into a cacheable back
|
||||||
|
buffer, presents only the damaged region, and is driven over IPC by the `runtime.display`
|
||||||
|
client — proven end-to-end by a separate demo process. Two limitations are deliberate and
|
||||||
|
documented (docs/display.md): no runtime mode-setting (native backend) and no true vsync
|
||||||
|
(no vblank on a dumb framebuffer). Next steps are the Deferred items below.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Deferred (explicitly not in this plan)
|
||||||
|
|
||||||
|
- **Shared-memory surfaces** — generalize M13 capability passing to memory objects
|
||||||
|
(`shm_create`/`shm_map`), so bitmap clients hand the compositor a rendered surface
|
||||||
|
instead of drawing commands. The compositor's layer model already anticipates it.
|
||||||
|
- **Native backend (Bochs DISPI, then virtio-gpu)** — behind the same internal backend
|
||||||
|
interface as the dumb framebuffer: EDID mode list + runtime resolution/bpp change +
|
||||||
|
(eventually) a vblank/flip path for true vsync.
|
||||||
|
- **Driver/compositor process split** — only when a second backend or a second head makes
|
||||||
|
the abstraction pay for itself.
|
||||||
@@ -0,0 +1,173 @@
|
|||||||
|
# Display v2 — build plan (pluggable scanout: GOP floor + virtio-gpu native)
|
||||||
|
|
||||||
|
The ordered, checkpointable build-out for [display-v2.md](display-v2.md). Each milestone
|
||||||
|
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||||
|
[display-plan.md](display-plan.md). Read display-v2.md first for the *why*.
|
||||||
|
|
||||||
|
## Locked decisions (do not relitigate)
|
||||||
|
|
||||||
|
- **First native backend = virtio-gpu** (VM standard: mode-set + present/flush + vsync).
|
||||||
|
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||||
|
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||||
|
ever," not a live fall-back after a reprogram.
|
||||||
|
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||||
|
client-surface path.
|
||||||
|
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||||
|
|
||||||
|
## Conventions
|
||||||
|
|
||||||
|
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||||
|
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||||
|
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||||
|
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||||
|
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||||
|
|
||||||
|
## How to verify along the way
|
||||||
|
|
||||||
|
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||||
|
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||||
|
own pixels back**: the scanout resource is CPU-visible RAM (shm-backed) and the back buffer
|
||||||
|
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||||
|
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||||
|
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||||
|
"it's on screen."
|
||||||
|
|
||||||
|
- `zig build test` — host unit tests (backend selection, virtio struct sizes/encodings,
|
||||||
|
pixel-check helpers).
|
||||||
|
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial markers.
|
||||||
|
The virtio cases boot with `-device virtio-gpu` (a per-case `qemu_extra`).
|
||||||
|
- `run-x86-64` renders to a window — for the human's own satisfaction, **not** a gate.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## V1 — The scanout backend seam (refactor, no behaviour change) ✅
|
||||||
|
|
||||||
|
Extract scanout from the compositor so today's path becomes one backend among future ones.
|
||||||
|
|
||||||
|
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||||
|
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||||
|
(`canModeSet`/`hasVsync`, both false for GOP).
|
||||||
|
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||||
|
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||||
|
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||||
|
framebuffer geometry left in the compositor core.
|
||||||
|
- [x] The selection decision is the pure `chooseKind(native_available)` (gop unless a
|
||||||
|
native driver announced), split from the syscall-bound `select()`/`Gop.init()`.
|
||||||
|
|
||||||
|
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||||
|
is the only backend), and `zig build test` stays green.
|
||||||
|
|
||||||
|
## V2 — The `shm` cross-process memory capability (kernel) ✅
|
||||||
|
|
||||||
|
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||||
|
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||||
|
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||||
|
handle, maps them into the caller's shm arena → returns vaddr + handle; `shm_map(cap)`
|
||||||
|
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||||
|
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||||
|
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||||
|
dispatch by kind, so an `ShmObject` rides an `ipc_call` `send_cap` exactly like an
|
||||||
|
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||||
|
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||||
|
frames — the object owns them.
|
||||||
|
- [x] `library/runtime/shm.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||||
|
`map(handle) -> ptr`.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py shm` — `shm-client` creates a region, writes a
|
||||||
|
pattern, and passes its capability to `shm-server` as an `ipc_call` send_cap; the server
|
||||||
|
`shm_map`s it and reads the **same bytes** back → `shm: shared 4096 bytes ok`. Guardrail:
|
||||||
|
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||||
|
tests all still pass — the handle-table change broke no existing IPC.
|
||||||
|
|
||||||
|
## V3 — The virtio-gpu driver: bring-up + a frame on screen ✅
|
||||||
|
|
||||||
|
- [x] `system/drivers/virtio-gpu/`: claim the virtio-gpu PCI function (device-manager
|
||||||
|
match on the display/other class triple, driver self-confirms vendor 0x1AF4/device
|
||||||
|
0x1050 from config space), enable memory-space + bus-master, walk the vendor
|
||||||
|
capabilities in config space to find common-config + notify, map the BAR, negotiate
|
||||||
|
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||||
|
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||||
|
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||||
|
shm-shared surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||||
|
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||||
|
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||||
|
|
||||||
|
**Gate (met):** the `virtio-gpu` case (QEMU `-device virtio-gpu-pci`) boots the
|
||||||
|
device-manager stack, which discovers the function and spawns the driver; the driver writes
|
||||||
|
a known test pattern into the scanout backing, `transfer_to_host_2d` + `resource_flush`es
|
||||||
|
it, and **waits for the device's used-ring ack**, then reads the backing back and checks the
|
||||||
|
pattern — logging `virtio-gpu: scanout 640x480 online` and `virtio-gpu: flush acked, pixel
|
||||||
|
check ok`. That proves virtqueue + resource + attach + set_scanout + transfer + flush end to
|
||||||
|
end without a screenshot (the used-ring ack is the device confirming it consumed the frame).
|
||||||
|
|
||||||
|
## V4 — The native backend + hot-attach ✅
|
||||||
|
|
||||||
|
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared `shm` scanout surface
|
||||||
|
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||||
|
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||||
|
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||||
|
- [x] The driver **announces** to `.display` after bring-up (looks it up with a bounded retry,
|
||||||
|
sends `attach_scanout` with the geometry + the shared surface as an `ipc_call` send_cap).
|
||||||
|
The compositor maps it, looks up `.scanout` itself (no need to pass the endpoint — the
|
||||||
|
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||||
|
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||||
|
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||||
|
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shm_physical` (a
|
||||||
|
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||||
|
|
||||||
|
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||||
|
boots the whole system) starts the compositor + `display-demo` + device-manager; the driver
|
||||||
|
announces, the compositor logs `display: scanout upgraded to virtio-gpu`, drives frames through
|
||||||
|
the native backend, and **reads a pixel back** from the shared surface after a present to
|
||||||
|
confirm the composited frame landed (`display: native present verified`), while `display-demo:
|
||||||
|
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||||
|
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||||
|
|
||||||
|
## V5 — Mode-setting, EDID, and vsync ✅
|
||||||
|
|
||||||
|
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||||
|
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||||
|
resource + shared surface are sized to the largest mode, so `set_mode` just re-points the
|
||||||
|
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||||
|
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||||
|
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||||
|
fence when the frame is on screen, which the used-ring ack the synchronous present waits on
|
||||||
|
already gates — a tear-free present.
|
||||||
|
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasVsync` = true.
|
||||||
|
|
||||||
|
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||||
|
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||||
|
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||||
|
fenced present path is exercised and confirmed (`display: vsync present ok`) — both from serial,
|
||||||
|
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||||
|
|
||||||
|
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||||
|
|
||||||
|
- [x] The virtio-gpu driver now **hellos** the device manager (role: bus) so it is properly
|
||||||
|
supervised — no longer stopped at the hello deadline — and is restarted on death. On
|
||||||
|
driver loss the compositor keeps the last frame (its `.scanout` calls now return
|
||||||
|
`-EPEER` instead of hanging — a kernel fix: an endpoint is marked dead when its owner
|
||||||
|
dies) and **re-attaches** when the restarted driver re-announces. A permanent give-up
|
||||||
|
(crash-loop cap) leaves the frozen frame; GOP is not re-taken.
|
||||||
|
- [x] `test/qemu_test.py`: the `virtio-gpu`, `display-native` (hot-attach), `display-modeset`,
|
||||||
|
and `display-reattach` (driver-kill/re-attach) cases. display-v2.md status updated.
|
||||||
|
|
||||||
|
**Gate (met):** the `display-reattach` case — device-manager (in `test-scanout-restart` mode)
|
||||||
|
kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||||
|
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||||
|
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||||
|
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||||
|
`supervision`, `shm`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||||
|
`display-modeset`) pass; default `zig build` is clean.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Deferred (explicitly not in this plan)
|
||||||
|
|
||||||
|
- **Client-rendered surfaces** — now unblocked by the `shm` capability (V2): an app renders
|
||||||
|
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||||
|
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||||
|
slots behind the same interface if wanted.
|
||||||
|
- **Real-GPU (NVIDIA/AMD/Intel) drivers** — out of scope; those devices stay on the GOP
|
||||||
|
floor by design.
|
||||||
|
- **Hardware-accelerated compositing / multiple heads** — future.
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
# The display service v2: a pluggable scanout backend
|
||||||
|
|
||||||
|
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||||
|
virtio-gpu driver announces itself, hot-attaches a native backend over the shared `shm`
|
||||||
|
scanout surface — with runtime mode-setting, EDID, and fenced (vsync) presents, and it
|
||||||
|
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||||
|
|
||||||
|
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||||
|
composites a layer stack into a cacheable back buffer and streams damage to the linear
|
||||||
|
framebuffer the firmware handed over. That path is portable and good: it drives any GPU,
|
||||||
|
including a real NVIDIA card at an ultrawide's native resolution, with zero GPU-specific
|
||||||
|
code. v2 keeps it as the **floor** and makes *scanout* — how a finished frame reaches the
|
||||||
|
panel — a **pluggable backend**, so the compositor can **upgrade to a real GPU driver when
|
||||||
|
one is present** and fall back to the framebuffer when it isn't.
|
||||||
|
|
||||||
|
The compositor itself (layers, back buffer, damage) does not change. Only the last step —
|
||||||
|
"put this frame on screen" — becomes swappable.
|
||||||
|
|
||||||
|
## The shape
|
||||||
|
|
||||||
|
```
|
||||||
|
compositor (display service) ── layer stack + back buffer + damage (unchanged)
|
||||||
|
│ composites a frame, then: backend.present(damage)
|
||||||
|
▼
|
||||||
|
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||||
|
│
|
||||||
|
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||||
|
│ Always available. No mode-set, no vsync. THE FLOOR.
|
||||||
|
│
|
||||||
|
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||||
|
service: present via a shared resource + flush (real vsync),
|
||||||
|
EDID mode list, runtime mode-set.
|
||||||
|
```
|
||||||
|
|
||||||
|
A **backend** is a small interface the compositor calls:
|
||||||
|
|
||||||
|
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||||
|
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||||
|
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||||
|
a virtio flush, optionally vsync-fenced, for the native path),
|
||||||
|
- capability queries — `canModeSet`, `hasVsync` — and, when supported, `modes()` /
|
||||||
|
`setMode(m)`.
|
||||||
|
|
||||||
|
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||||
|
today; everything device-specific lives behind the interface.
|
||||||
|
|
||||||
|
## Selection and hot-attach
|
||||||
|
|
||||||
|
The choice is **dynamic**, because a GPU driver is spawned asynchronously (the device
|
||||||
|
manager brings it up after boot), and because danos is meant to be resilient:
|
||||||
|
|
||||||
|
1. **Boot on GOP.** The compositor starts on `GopBackend` immediately, so there is never a
|
||||||
|
blank screen while drivers load — the exact v1 behaviour.
|
||||||
|
2. **Upgrade on announce.** When the virtio-gpu driver has claimed its device and set up a
|
||||||
|
scanout, it **announces itself to the display service** (a `push`: the driver looks up
|
||||||
|
`.display` and sends an *attach-scanout* message carrying its `scanout` endpoint as a
|
||||||
|
capability). The compositor switches to `VirtioGpuBackend` and re-presents the current
|
||||||
|
frame full-screen. Push beats polling — the compositor doesn't know a priori which
|
||||||
|
driver, if any, exists, and danos has no service-registration pub/sub.
|
||||||
|
3. **Native is restartable, not fallback-on-crash.** Once a native driver has reprogrammed
|
||||||
|
the device, the firmware's GOP framebuffer is **stale** — "native → GOP" is not a clean
|
||||||
|
fall-back. So a native driver that **crashes** is *restarted* by its supervisor (the
|
||||||
|
resilience work already merged), re-announces, and the compositor **re-attaches**
|
||||||
|
(native → native). The screen freezes on the last frame during the gap — acceptable.
|
||||||
|
4. **GOP is the floor for "no driver was ever there."** On a real GPU (NVIDIA/AMD/Intel)
|
||||||
|
the class-0x03 device matches nothing in the driver table, no `scanout` is ever
|
||||||
|
announced, and the compositor stays on GOP forever — no special-casing. Only if a
|
||||||
|
native driver *permanently* gives up (crash-loop cap) does the compositor attempt GOP
|
||||||
|
again, and even then only if the LFB is still mappable.
|
||||||
|
|
||||||
|
## The shared-memory primitive this needs
|
||||||
|
|
||||||
|
virtio-gpu's scanout resource is **guest RAM** — the driver allocates it and attaches it
|
||||||
|
to a virtio resource, and the compositor composes into it. That means the compositor
|
||||||
|
writing into the driver's buffer is **cross-process memory sharing**, the primitive v1
|
||||||
|
deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural generalization
|
||||||
|
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||||
|
|
||||||
|
```
|
||||||
|
shm_create(len) -> {handle, vaddr} // a shareable, page-aligned RAM region
|
||||||
|
… pass `handle` as the send_cap on an ipc_call …
|
||||||
|
shm_map(cap) -> vaddr // the receiver maps the same physical pages
|
||||||
|
```
|
||||||
|
|
||||||
|
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||||
|
client-rendered surfaces (an app composing its own bitmap and handing the compositor a
|
||||||
|
reference instead of drawing by command). One piece of kernel work, two features.
|
||||||
|
|
||||||
|
## The virtio-gpu driver
|
||||||
|
|
||||||
|
A new ring-3 driver process (the topology v1 anticipated — "split the driver from the
|
||||||
|
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||||
|
|
||||||
|
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||||
|
- creates a **2D scanout resource** backed by an `shm` region, `attach_backing`s it,
|
||||||
|
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||||
|
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||||
|
at a chosen mode for **runtime mode-setting**,
|
||||||
|
- registers a `scanout` service and announces to the display service.
|
||||||
|
|
||||||
|
Its `resource_flush` is the real **present** — and gives a genuine **vsync/tear-free**
|
||||||
|
path a dumb GOP framebuffer can't.
|
||||||
|
|
||||||
|
## What v2 unlocks — and its honest scope
|
||||||
|
|
||||||
|
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||||
|
refresh / bpp), **EDID** enumeration, and **vsync**. But only on devices we have a driver
|
||||||
|
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||||
|
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||||
|
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||||
|
the **pluggable architecture** (a driver slots in when one exists) and a **rich, vsync'd
|
||||||
|
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||||
|
|
||||||
|
## Locked decisions
|
||||||
|
|
||||||
|
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||||
|
present/flush (and vsync), and exercises the whole pluggable design. Tested with QEMU
|
||||||
|
`-device virtio-gpu`.
|
||||||
|
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||||
|
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||||
|
fall-back after a reprogram.
|
||||||
|
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||||
|
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||||
|
client-surface path.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||||
|
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||||
|
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||||
|
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||||
+264
@@ -0,0 +1,264 @@
|
|||||||
|
# The display service: a framebuffer compositor
|
||||||
|
|
||||||
|
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||||
|
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||||
|
into it directly. That console is a stop-gap. The **display service**
|
||||||
|
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||||
|
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||||
|
**presents** finished frames to the screen — the display half of the GUI track
|
||||||
|
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||||
|
|
||||||
|
This note is the architecture and the reasoning behind it. The concrete build order
|
||||||
|
lives in [display-plan.md](display-plan.md).
|
||||||
|
|
||||||
|
## First, a distinction that shapes everything: GOP vs. the PCI device
|
||||||
|
|
||||||
|
It is tempting to think "the GOP framebuffer" and "the VGA-compatible display
|
||||||
|
controller in the PCIe tree" are two different things. They are not — they are **two
|
||||||
|
interfaces to the same silicon, at different times and different levels**, and knowing
|
||||||
|
which one you're holding decides what you can do.
|
||||||
|
|
||||||
|
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||||
|
linear framebuffer pointer and can set video modes — but only until
|
||||||
|
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||||
|
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||||
|
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||||
|
mode list, no EDID. What survives is the frozen snapshot in
|
||||||
|
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||||
|
pitch, format}`, and nothing more.
|
||||||
|
|
||||||
|
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||||
|
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||||
|
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||||
|
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||||
|
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||||
|
VRAM BAR. danos already decodes this device
|
||||||
|
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||||
|
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||||
|
triple) — but nothing binds it yet.
|
||||||
|
|
||||||
|
What that difference costs you, concretely:
|
||||||
|
|
||||||
|
| You want to… | Dumb GOP framebuffer (boot handoff) | Native device driver (PCI 0x03) |
|
||||||
|
|-------------------------------------------|-------------------------------------|------------------------------------------|
|
||||||
|
| **Report** the current mode | ✅ from the handoff | ✅ |
|
||||||
|
| **Change resolution / bpp at runtime** | ❌ GOP is gone | ✅ program DISPI regs / virtio-gpu queue |
|
||||||
|
| **Re-read EDID, enumerate monitor modes** | ❌ | ✅ the device exposes an EDID block |
|
||||||
|
| **Refresh rate** | ❌ (virtual anyway) | only a real KMS driver — far future |
|
||||||
|
| **vblank / tear-free present** | ❌ no vblank signal | ✅ vblank IRQ + page-flip (real GPUs) |
|
||||||
|
| **Works on the Pi (no PCI VGA)** | ✅ VideoCore hands a simple FB | ✗ per-device |
|
||||||
|
|
||||||
|
The lesson: the **portable base for the whole GUI stack is the GOP / boot-handoff linear
|
||||||
|
framebuffer**. Runtime mode-setting is a *per-device upgrade* layered on top — and on
|
||||||
|
the Raspberry Pis there is no PCI VGA at all, so the neutral framebuffer is the only
|
||||||
|
thing all three target machines share. That is why the display service is built on the
|
||||||
|
dumb framebuffer first, with the native backend as an optional module behind the same
|
||||||
|
interface.
|
||||||
|
|
||||||
|
## Two constraints this service exists to meet
|
||||||
|
|
||||||
|
Like the input service — which existed partly to motivate the asynchronous
|
||||||
|
[`ipc_send`](ipc.md) primitive — the display service runs straight into two limits the
|
||||||
|
rest of the system hasn't had to face:
|
||||||
|
|
||||||
|
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||||
|
mapped into the kernel's physmap, and is touched only by
|
||||||
|
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||||
|
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||||
|
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||||
|
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||||
|
touch the pixels**. (See "The handoff" below — this is built.)
|
||||||
|
|
||||||
|
2. **danos has no cross-process shared memory.** The memory syscalls are `mmap`
|
||||||
|
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||||
|
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||||
|
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||||
|
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||||
|
use it — it would have to *map* another process's memory, which nothing allows. This
|
||||||
|
is deferred (see "What v1 does not do"), because v1 sidesteps it entirely.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ── owns the boot framebuffer; bootstrap console only
|
||||||
|
│ seeds a "display0" device node from BootInformation.framebuffer
|
||||||
|
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||||
|
│ plus DisplayInfo{width, height, pitch, format})
|
||||||
|
▼
|
||||||
|
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||||
|
│ device.claim(display0) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||||
|
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||||
|
│ owns: an ordered LAYER STACK + a per-frame DAMAGE list
|
||||||
|
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||||
|
│ backend is an INTERNAL interface: {gop-fb} today; {bochs-dispi, virtio-gpu} later
|
||||||
|
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||||
|
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||||
|
drawing clients (v1) surface clients (deferred)
|
||||||
|
runtime.display commands: runtime.display surfaces:
|
||||||
|
create_layer / configure_layer shm_create → pass as a capability →
|
||||||
|
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||||
|
present client-rendered bitmap directly
|
||||||
|
```
|
||||||
|
|
||||||
|
The bring-up sequence mirrors a hardware driver's — it is the
|
||||||
|
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||||
|
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||||
|
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||||
|
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||||
|
`extern struct` messages and an `Operation` tag).
|
||||||
|
|
||||||
|
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||||
|
composites — it does not split a "framebuffer driver" from a "compositor" the way input
|
||||||
|
splits `ps2-bus` from the input service. The backend (dumb FB vs. a native GPU) is an
|
||||||
|
*internal* interface, not a process boundary. That boundary earns its keep only when a
|
||||||
|
second backend or a second monitor appears; until then it is complexity with no payoff.
|
||||||
|
|
||||||
|
## The handoff: a device node + a write-combining map
|
||||||
|
|
||||||
|
The framebuffer crosses into user space through the machinery that already exists for
|
||||||
|
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||||
|
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||||
|
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||||
|
re-claims it).
|
||||||
|
|
||||||
|
- The kernel seeds a synthetic **`display0`** node into the
|
||||||
|
[devices-broker](../system/kernel/devices-broker.zig) at init, from
|
||||||
|
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||||
|
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||||
|
`DisplayInfo{width, height, pitch, format}` (the memory resource says *where* and *how
|
||||||
|
big*; `DisplayInfo` says how to *interpret* the bytes).
|
||||||
|
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||||
|
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||||
|
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||||
|
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||||
|
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||||
|
back→front blit unusably slow.
|
||||||
|
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||||
|
LFB. A panic is the one exception — by then the service is likely dead anyway, and a
|
||||||
|
panic on screen wins.
|
||||||
|
|
||||||
|
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||||
|
`vfs`/`input`/`device-manager` ([init.zig](../system/services/init/init.zig)), and it
|
||||||
|
self-discovers `display0` with `device.enumerate`. The [device manager](device-manager.md)
|
||||||
|
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||||
|
this singleton synthetic node.
|
||||||
|
|
||||||
|
## Double buffering and the write-combining discipline
|
||||||
|
|
||||||
|
Two buffers, with deliberately different memory types:
|
||||||
|
|
||||||
|
- The **front buffer** is the LFB — **write-combining**: fast to *write*, slow to
|
||||||
|
*read*. The rule is therefore **never read the front buffer**. Only ever stream into
|
||||||
|
it, sequentially.
|
||||||
|
- The **back buffer** is ordinary **cacheable** RAM (`mmap`), the same geometry. All
|
||||||
|
compositing happens here, where reads and read-modify-write blends are cheap.
|
||||||
|
|
||||||
|
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||||
|
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||||
|
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||||
|
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||||
|
|
||||||
|
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||||
|
|
||||||
|
These are two different artifacts, and the dumb framebuffer fixes exactly one of them:
|
||||||
|
|
||||||
|
- **Flicker** is the user seeing intermediate, half-drawn states (a clear-then-redraw
|
||||||
|
flash). Double buffering **eliminates it completely** — the screen only ever receives
|
||||||
|
whole, finished frames.
|
||||||
|
- **Tearing** is a present landing while the display's scanout beam is mid-frame, so the
|
||||||
|
top of the screen shows the new frame and the bottom the old. Avoiding it requires
|
||||||
|
presenting during the vertical blank (**vsync**) — which needs a vblank signal. **A
|
||||||
|
dumb GOP framebuffer has no vblank.**
|
||||||
|
|
||||||
|
So v1 is **flicker-free**, and it *minimizes* the tear window by presenting only damaged
|
||||||
|
rectangles (less to copy → a smaller window in which the beam can catch a half-updated
|
||||||
|
frame), but it is **not tear-free**. Genuine vsync waits for a backend with a vblank IRQ
|
||||||
|
or a flush/flip path — a native-device capability, not something the firmware
|
||||||
|
framebuffer can offer. Stated plainly here so the limitation is understood, not
|
||||||
|
discovered.
|
||||||
|
|
||||||
|
## Layers and the client protocol
|
||||||
|
|
||||||
|
The compositor holds an **ordered stack of layers**. Each layer has a rectangle, a
|
||||||
|
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||||
|
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||||
|
|
||||||
|
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||||
|
immediate-mode command protocol — essentially the model early X used, and enough for a
|
||||||
|
shell, a terminal, a cursor, and a wallpaper:
|
||||||
|
|
||||||
|
| Operation | Meaning |
|
||||||
|
|--------------------|---------------------------------------------------------------|
|
||||||
|
| `info` | report `{width, height, pitch, format}` of the display |
|
||||||
|
| `create_layer` | allocate a server-owned surface, return a layer handle |
|
||||||
|
| `configure_layer` | set a layer's rect, z-order, visibility |
|
||||||
|
| `destroy_layer` | release a layer |
|
||||||
|
| `fill_rect` | fill a rectangle of a layer with a colour |
|
||||||
|
| `blit_tile` | copy a small client-supplied pixel tile into a layer (inline) |
|
||||||
|
| `damage` | mark a region of a layer dirty |
|
||||||
|
| `present` | composite dirty layers and flush to the screen |
|
||||||
|
|
||||||
|
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||||
|
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||||
|
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||||
|
policy in the client.
|
||||||
|
|
||||||
|
## `runtime.display`
|
||||||
|
|
||||||
|
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||||
|
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||||
|
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||||
|
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||||
|
runtime, as with every other danos service.
|
||||||
|
|
||||||
|
## What v1 does not do (and why that's fine)
|
||||||
|
|
||||||
|
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
||||||
|
both are clean additions behind the interfaces v1 establishes.
|
||||||
|
|
||||||
|
- **Client-rendered surfaces (shared memory).** The fast path for a bitmap-heavy app is
|
||||||
|
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||||
|
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||||
|
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||||
|
from *endpoints* to *memory objects* (`shm_create(len) → {cap, vaddr}`, pass `cap` on
|
||||||
|
an `ipc_call`, receiver `shm_map(cap) → vaddr`). v1 avoids it because server-owned
|
||||||
|
surfaces already prove the whole pipeline.
|
||||||
|
|
||||||
|
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||||
|
resolution / bpp at runtime needs the raw PCI device. The first native backend is
|
||||||
|
Bochs DISPI — the register interface QEMU's `-device VGA` exposes — behind the same
|
||||||
|
internal backend interface the dumb framebuffer sits behind. Refresh-rate and colour
|
||||||
|
management (a gamma LUT) are real-GPU-KMS territory, far beyond this.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
Three QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||||
|
test/qemu_test.py <case>`), each layering on the last:
|
||||||
|
|
||||||
|
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||||
|
the claim → `mmio_map` leaf is genuinely **write-combining** (PAT entry 4), asserted at
|
||||||
|
the page-table level.
|
||||||
|
- **`display-service`** — the compositor comes up: it claims the framebuffer, allocates
|
||||||
|
the cacheable back buffer, presents a cleared frame through the double-buffer path
|
||||||
|
(`display: online … / presented frame 0`), and a startup **self-check** composites two
|
||||||
|
overlapping layers on the real framebuffer and reads them back — overlap = the top
|
||||||
|
layer — logging `display: compositor self-check ok`.
|
||||||
|
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||||
|
[`display-demo`](../system/services/display-demo/) client (the
|
||||||
|
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper, a
|
||||||
|
sliding rectangle, a cursor — through the layer client API and heartbeats
|
||||||
|
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||||
|
the [input test](input.md) proves an event travels source → service → subscriber. The
|
||||||
|
visible motion itself is a screenshot away via `zig build run-x86-64`.
|
||||||
|
|
||||||
|
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||||
|
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||||
|
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||||
|
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||||
|
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||||
|
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||||
|
- [display-plan.md](display-plan.md) — the ordered build-out.
|
||||||
@@ -27,6 +27,15 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
|||||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||||
a new architecture's UART is what makes the same tests run there.
|
a new architecture's UART is what makes the same tests run there.
|
||||||
|
|
||||||
|
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||||
|
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||||
|
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||||
|
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||||
|
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||||
|
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||||
|
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||||
|
so a serial-enabled image is still safe on real hardware.)
|
||||||
|
|
||||||
## In-kernel test cases
|
## In-kernel test cases
|
||||||
|
|
||||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||||
|
|||||||
@@ -38,6 +38,13 @@ pub const Device = struct {
|
|||||||
return self.transfer(.write, lba, count, physical);
|
return self.transfer(.write, lba, count, physical);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||||
|
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||||
|
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||||
|
pub fn flush(self: Device) bool {
|
||||||
|
return self.transfer(.flush, 0, 0, 0);
|
||||||
|
}
|
||||||
|
|
||||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||||
var reply: [protocol.reply_size]u8 = undefined;
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
|||||||
@@ -0,0 +1,198 @@
|
|||||||
|
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||||
|
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||||
|
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
||||||
|
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("display-protocol");
|
||||||
|
|
||||||
|
/// The display's current mode, as `info()` reports it.
|
||||||
|
pub const Info = struct {
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||||
|
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The service endpoint, looked up once and cached.
|
||||||
|
var handle: ?ipc.Handle = null;
|
||||||
|
|
||||||
|
/// Look up the display service, retrying while it comes up (a client races its
|
||||||
|
/// registration at boot). Returns the endpoint, or null if it never appears.
|
||||||
|
fn service() ?ipc.Handle {
|
||||||
|
if (handle) |h| return h;
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.display)) |h| {
|
||||||
|
handle = h;
|
||||||
|
return h;
|
||||||
|
}
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||||
|
/// so callers can read `info`/`layer` fields on success.
|
||||||
|
fn transact(request: protocol.Request, out: *protocol.Reply) bool {
|
||||||
|
const h = service() orelse return false;
|
||||||
|
var req = request;
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
out.* = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||||
|
return out.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The display's current mode, or null if the service never came up.
|
||||||
|
pub fn info() ?Info {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
if (!transact(.{ .operation = @intFromEnum(protocol.Operation.info) }, &reply)) return null;
|
||||||
|
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Composite the dirty layers and flush the frame to the screen.
|
||||||
|
pub fn present() bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One selectable display mode.
|
||||||
|
pub const Mode = protocol.Mode;
|
||||||
|
|
||||||
|
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||||
|
/// (zero on the GOP floor, or if the service never came up).
|
||||||
|
pub fn modes(out: []Mode) usize {
|
||||||
|
const h = service() orelse return 0;
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||||
|
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||||
|
if (len < protocol.modes_reply_size) return 0;
|
||||||
|
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||||
|
if (answer.status != 0) return 0;
|
||||||
|
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||||
|
for (0..count) |i| out[i] = answer.modes[i];
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||||
|
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||||
|
pub fn setMode(width: u32, height: u32) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||||
|
if (changed) mode = null; // the cached mode is stale now
|
||||||
|
return changed;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||||
|
var mode: ?Info = null;
|
||||||
|
|
||||||
|
fn cachedInfo() ?Info {
|
||||||
|
if (mode) |m| return m;
|
||||||
|
const i = info() orelse return null;
|
||||||
|
mode = i;
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||||
|
/// client packs colours through this so it never has to know the byte order itself.
|
||||||
|
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||||
|
const format = if (cachedInfo()) |i| i.format else 0;
|
||||||
|
return protocol.pack(format, r, g, b);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||||
|
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||||
|
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||||
|
pub const Layer = struct {
|
||||||
|
id: u32,
|
||||||
|
|
||||||
|
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||||
|
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.fill_rect),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
.colour = colour,
|
||||||
|
}, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||||
|
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||||
|
/// `protocol.maximum_payload`.
|
||||||
|
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||||
|
var request = protocol.Request{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.blit_tile),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
};
|
||||||
|
const header = std.mem.asBytes(&request);
|
||||||
|
if (header.len + pixels.len > protocol.message_maximum) return false;
|
||||||
|
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(buffer[0..header.len], header);
|
||||||
|
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||||
|
const h_svc = service() orelse return false;
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Move / restack / show or hide the layer.
|
||||||
|
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.configure_layer),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.z = z,
|
||||||
|
.visible = if (visible) 1 else 0,
|
||||||
|
}, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||||
|
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||||
|
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.damage),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
}, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the layer and its surface.
|
||||||
|
pub fn destroy(self: Layer) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{ .operation = @intFromEnum(protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||||
|
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||||
|
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
if (!transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.create_layer),
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
.z = z,
|
||||||
|
.visible = 1,
|
||||||
|
}, &reply)) return null;
|
||||||
|
return .{ .id = reply.layer };
|
||||||
|
}
|
||||||
@@ -38,6 +38,11 @@ pub const device = @import("device.zig");
|
|||||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||||
pub const dma = @import("dma.zig");
|
pub const dma = @import("dma.zig");
|
||||||
|
|
||||||
|
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||||
|
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shm.zig
|
||||||
|
/// and docs/display-v2.md.
|
||||||
|
pub const shm = @import("shm.zig");
|
||||||
|
|
||||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||||
pub const usb = @import("usb.zig");
|
pub const usb = @import("usb.zig");
|
||||||
@@ -46,6 +51,15 @@ pub const usb = @import("usb.zig");
|
|||||||
/// usb-storage). See library/runtime/block.zig.
|
/// usb-storage). See library/runtime/block.zig.
|
||||||
pub const block = @import("block.zig");
|
pub const block = @import("block.zig");
|
||||||
|
|
||||||
|
/// Display-service client: query the mode, and (from D3) create layers, draw, and
|
||||||
|
/// present frames. See library/runtime/display.zig and system/services/display/.
|
||||||
|
pub const display = @import("display.zig");
|
||||||
|
/// The display wire protocol (shared with the display service and its clients).
|
||||||
|
pub const display_protocol = @import("display-protocol");
|
||||||
|
/// The scanout wire protocol: the compositor's present channel to a native scanout driver
|
||||||
|
/// (virtio-gpu). See system/services/display/scanout-protocol.zig and docs/display-v2.md.
|
||||||
|
pub const scanout_protocol = @import("scanout-protocol");
|
||||||
|
|
||||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||||
/// layer danos programs use directly, and where the operations that later become
|
/// layer danos programs use directly, and where the operations that later become
|
||||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||||
|
|||||||
@@ -0,0 +1,57 @@
|
|||||||
|
//! User-space shared memory: `shm_create` / `shm_map`. A process creates a shareable,
|
||||||
|
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||||
|
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||||
|
//! `shm_map`s it to map the same physical pages. The kernel primitive under the display
|
||||||
|
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||||
|
//! generalization of capability passing from endpoints to memory objects.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||||
|
/// another process as an `ipc_call` send_cap.
|
||||||
|
pub const Region = struct {
|
||||||
|
ptr: [*]u8,
|
||||||
|
handle: ipc.Handle,
|
||||||
|
len: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||||
|
/// Returns the region or null on failure. Two return values — vaddr in rax, handle in rdx —
|
||||||
|
/// so this is a hand-written stub like `dma.alloc`.
|
||||||
|
pub fn create(len: usize) ?Region {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined; // out: the capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shm_create)),
|
||||||
|
[a0] "{rdi}" (len),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map the shared region named by a capability `handle` this process received (via an
|
||||||
|
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||||
|
/// Returns the pointer, or null on failure.
|
||||||
|
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||||
|
const r = sc.systemCall1(.shm_map, handle);
|
||||||
|
if (failed(r)) return null;
|
||||||
|
return @ptrFromInt(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||||
|
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||||
|
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||||
|
/// `attach_backing`. Returns null on failure.
|
||||||
|
pub fn physical(handle: ipc.Handle) ?usize {
|
||||||
|
const r = sc.systemCall1(.shm_physical, handle);
|
||||||
|
if (failed(r)) return null;
|
||||||
|
return r;
|
||||||
|
}
|
||||||
@@ -60,6 +60,9 @@ pub const SystemCall = enum(u64) {
|
|||||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||||
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||||
|
shm_create = 34, // shm_create(len) -> vaddr (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||||
|
shm_map = 35, // shm_map(cap) -> vaddr: map the shared region named by a received capability into this AS (the same physical pages the creator sees)
|
||||||
|
shm_physical = 36, // shm_physical(cap) -> paddr: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -183,6 +186,9 @@ pub const ServiceId = enum(u32) {
|
|||||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||||
|
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||||
|
shm_test = 10, // the shm test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||||
|
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+28
-62
@@ -3,11 +3,13 @@
|
|||||||
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||||
//! bootloader handed us) and translates the static tables into the generic
|
//! bootloader handed us) and translates the static tables into the generic
|
||||||
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||||
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
//! source. This is deliberately the *static-table* path, and **only** that: MADT
|
||||||
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
|
||||||
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
|
||||||
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
//! interpreted here — the kernel collects the blobs and publishes them on the
|
||||||
//! interpretation is a separate, larger subproject.
|
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
|
||||||
|
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
|
||||||
|
//! off the single-core critical path (nothing else runs alongside it there).
|
||||||
//!
|
//!
|
||||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||||
@@ -19,7 +21,6 @@ const boot_handoff = @import("boot-handoff");
|
|||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const parameters = @import("parameters");
|
const parameters = @import("parameters");
|
||||||
const device_model = @import("device-model.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const aml = @import("aml/aml.zig");
|
|
||||||
const DeviceTree = device_model.DeviceTree;
|
const DeviceTree = device_model.DeviceTree;
|
||||||
const Hal = device_model.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
@@ -37,8 +38,11 @@ pub const RegisterAccess = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
/// The power register map, extracted from the FADT during discovery. Populated by
|
||||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
|
||||||
|
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
|
||||||
|
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
|
||||||
|
/// writes the PM1 control register itself). So this holds only the FADT scalars.
|
||||||
pub const PowerInformation = struct {
|
pub const PowerInformation = struct {
|
||||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||||
@@ -54,10 +58,6 @@ pub const PowerInformation = struct {
|
|||||||
reset: RegisterAccess = .{},
|
reset: RegisterAccess = .{},
|
||||||
reset_value: u8 = 0,
|
reset_value: u8 = 0,
|
||||||
reset_supported: bool = false,
|
reset_supported: bool = false,
|
||||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
|
||||||
/// packages.
|
|
||||||
s5: ?aml.SleepType = null,
|
|
||||||
s3: ?aml.SleepType = null,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||||
@@ -141,19 +141,6 @@ const maximum_cpus = parameters.maximum_cpus;
|
|||||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||||
pub var cpu_information: CpuInformation = .{};
|
pub var cpu_information: CpuInformation = .{};
|
||||||
|
|
||||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
|
||||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
|
||||||
pub const AmlStats = struct {
|
|
||||||
nodes: usize = 0,
|
|
||||||
consumed: usize = 0,
|
|
||||||
total: usize = 0,
|
|
||||||
};
|
|
||||||
pub var aml_stats: AmlStats = .{};
|
|
||||||
|
|
||||||
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
|
||||||
/// device enumeration later. Null until `discover` runs successfully.
|
|
||||||
pub var namespace: ?aml.Namespace = null;
|
|
||||||
|
|
||||||
/// Physical address of the DSDT the FADT points at, or 0.
|
/// Physical address of the DSDT the FADT points at, or 0.
|
||||||
pub var dsdt_physical: u64 = 0;
|
pub var dsdt_physical: u64 = 0;
|
||||||
|
|
||||||
@@ -165,8 +152,9 @@ var fadt_physical: u64 = 0;
|
|||||||
var fadt_length: u64 = 0;
|
var fadt_length: u64 = 0;
|
||||||
|
|
||||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
// address + length of each table's post-header bytecode. The kernel does not
|
||||||
// for the sleep-state (`_Sx`) packages.
|
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
|
||||||
|
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
|
||||||
var aml_block_physical: [32]u64 = undefined;
|
var aml_block_physical: [32]u64 = undefined;
|
||||||
var aml_block_len: [32]usize = undefined;
|
var aml_block_len: [32]usize = undefined;
|
||||||
var aml_block_count: usize = 0;
|
var aml_block_count: usize = 0;
|
||||||
@@ -373,7 +361,7 @@ const Hpet = extern struct {
|
|||||||
|
|
||||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
|
||||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
if (rsdp_physical == 0) return error.NoRsdp;
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
boot_memory_regions = memory_regions;
|
boot_memory_regions = memory_regions;
|
||||||
@@ -383,8 +371,6 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
fadt_physical = 0;
|
fadt_physical = 0;
|
||||||
fadt_length = 0;
|
fadt_length = 0;
|
||||||
platform_information = .{};
|
platform_information = .{};
|
||||||
aml_stats = .{};
|
|
||||||
namespace = null;
|
|
||||||
dsdt_physical = 0;
|
dsdt_physical = 0;
|
||||||
aml_block_count = 0;
|
aml_block_count = 0;
|
||||||
|
|
||||||
@@ -401,34 +387,20 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
|
||||||
// read the sleep types from it.
|
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
|
||||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
// and the power register map. The AML bytecode (device enumeration and the
|
||||||
for (0..aml_block_count) |i| {
|
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
|
||||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
// job: it claims the acpi-tables node published below, parses the same blobs,
|
||||||
}
|
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
|
||||||
const active = blocks[0..aml_block_count];
|
// AML in the kernel keeps boot latency off the critical, single-core path.
|
||||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
|
||||||
namespace = pr.namespace;
|
|
||||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
|
||||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
|
||||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
|
||||||
// The namespace's Device objects are no longer folded into the kernel
|
|
||||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
|
||||||
// (published below), re-parses the same blobs, and registers + reports
|
|
||||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
|
||||||
// \_S5 sleep type above.
|
|
||||||
} else |_| {
|
|
||||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
|
||||||
// whatever the FADT alone provided.
|
|
||||||
}
|
|
||||||
|
|
||||||
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
// claimant. Kept even when the kernel-side device building (above) retires
|
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
|
||||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
// now that the kernel keeps only the *static* tables for itself.
|
||||||
publishAcpiTablesNode(device_tree) catch {};
|
publishAcpiTablesNode(device_tree) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -461,12 +433,6 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
|||||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
|
||||||
pub fn amlDeviceCount() usize {
|
|
||||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
@@ -502,7 +468,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
|||||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||||
parseDmar(hal, header);
|
parseDmar(hal, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||||
addAmlBlock(sdt_physical);
|
addAmlBlock(sdt_physical);
|
||||||
}
|
}
|
||||||
// Any other signature is recognised but left opaque for now.
|
// Any other signature is recognised but left opaque for now.
|
||||||
@@ -740,8 +706,8 @@ const fadt_x_pm_tmr_blk = 208; // GAS
|
|||||||
const flag_reset_register_supported = 1 << 10;
|
const flag_reset_register_supported = 1 << 10;
|
||||||
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||||
|
|
||||||
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
/// FADT -> the power register map (into `power_information`) and the DSDT address,
|
||||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
|
||||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
const len: usize = header.length;
|
const len: usize = header.length;
|
||||||
|
|||||||
@@ -38,6 +38,13 @@ pub const DeviceClass = enum(u32) {
|
|||||||
/// resources; the (class, subclass, protocol) triple that says what it is
|
/// resources; the (class, subclass, protocol) triple that says what it is
|
||||||
/// travels in the bus report's identity, not here.
|
/// travels in the bus report's identity, not here.
|
||||||
usb_device,
|
usb_device,
|
||||||
|
/// A scanout framebuffer: a linear region of pixel memory the display service
|
||||||
|
/// claims and maps. Unlike the other classes this one is not firmware-discovered
|
||||||
|
/// — the kernel seeds it from the loader's [[boot-handoff]] framebuffer
|
||||||
|
/// (`devices_broker.seedDisplay`). Its one `memory` resource is the framebuffer,
|
||||||
|
/// flagged write-combining; the geometry to interpret it travels in
|
||||||
|
/// `DeviceDescriptor.display`.
|
||||||
|
display,
|
||||||
unknown,
|
unknown,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -59,10 +66,39 @@ pub const ResourceDescriptor = extern struct {
|
|||||||
kind: u64, // a ResourceKind value
|
kind: u64, // a ResourceKind value
|
||||||
start: u64,
|
start: u64,
|
||||||
len: u64,
|
len: u64,
|
||||||
|
/// A bitmask of `resource_flag_*` hints. Zero for a plain register/RAM window;
|
||||||
|
/// the kernel reads it when it maps the resource. Defaulted so every existing
|
||||||
|
/// literal (which never set flags) keeps compiling and lays out identically.
|
||||||
|
flags: u64 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// `ResourceDescriptor.flags`: map this `memory` resource **write-combining** rather
|
||||||
|
/// than strong-uncacheable — for a framebuffer, where batched bursts to pixel memory
|
||||||
|
/// are the whole point (an uncacheable framebuffer blit is glacial). See
|
||||||
|
/// `mmio_map` (system/kernel/process.zig) and `setupPat` (…/x86_64/paging.zig).
|
||||||
|
pub const resource_flag_write_combining: u64 = 1 << 0;
|
||||||
|
|
||||||
pub const maximum_device_resources = 8;
|
pub const maximum_device_resources = 8;
|
||||||
|
|
||||||
|
/// The byte order of a display's pixels — mirrors the loader's `PixelFormat`
|
||||||
|
/// ([[boot-handoff]]) with the same numeric values, but lives here so user space
|
||||||
|
/// (which must never import the loader↔kernel handoff) can name it. Only the two
|
||||||
|
/// linear 32-bpp layouts a console can paint into exist; see docs/gop.md.
|
||||||
|
pub const DisplayFormat = enum(u32) {
|
||||||
|
rgbx = 0, // byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved
|
||||||
|
bgrx = 1, // byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The geometry of a `display` device's framebuffer, carried in its descriptor so a
|
||||||
|
/// claiming driver knows how to interpret the pixel bytes its `memory` resource maps.
|
||||||
|
/// `pitch` is bytes per row (may exceed `width * 4`; see docs/framebuffer.md).
|
||||||
|
pub const DisplayInfo = extern struct {
|
||||||
|
width: u32 = 0, // visible pixels per row
|
||||||
|
height: u32 = 0, // visible rows
|
||||||
|
pitch: u32 = 0, // bytes from one row's start to the next
|
||||||
|
format: u32 = 0, // a DisplayFormat value
|
||||||
|
};
|
||||||
|
|
||||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||||
pub const no_parent: u64 = ~@as(u64, 0);
|
pub const no_parent: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
@@ -92,4 +128,9 @@ pub const DeviceDescriptor = extern struct {
|
|||||||
resource_count: u64,
|
resource_count: u64,
|
||||||
hid: [8]u8,
|
hid: [8]u8,
|
||||||
resources: [maximum_device_resources]ResourceDescriptor,
|
resources: [maximum_device_resources]ResourceDescriptor,
|
||||||
|
// Framebuffer geometry, meaningful only when `class` is `DeviceClass.display`
|
||||||
|
// (zeroed otherwise). Kept here — a class-specific field on the shared descriptor —
|
||||||
|
// the same way `pci_class` is meaningful only for `pci_device` and `hid` only for
|
||||||
|
// `acpi_device`.
|
||||||
|
display: DisplayInfo = .{},
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -22,13 +22,14 @@ pub const Resource = device_model.Resource;
|
|||||||
pub const ResourceKind = device_model.ResourceKind;
|
pub const ResourceKind = device_model.ResourceKind;
|
||||||
pub const Hal = device_model.Hal;
|
pub const Hal = device_model.Hal;
|
||||||
pub const PowerInformation = acpi.PowerInformation;
|
pub const PowerInformation = acpi.PowerInformation;
|
||||||
pub const AmlStats = acpi.AmlStats;
|
|
||||||
pub const PlatformInformation = acpi.PlatformInformation;
|
pub const PlatformInformation = acpi.PlatformInformation;
|
||||||
pub const RegisterAccess = acpi.RegisterAccess;
|
pub const RegisterAccess = acpi.RegisterAccess;
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
pub const IsoEntry = acpi.IsoEntry;
|
||||||
pub const Cpu = acpi.Cpu;
|
pub const Cpu = acpi.Cpu;
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||||
|
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||||
|
/// owned by the ring-3 acpi service), so they are not here.
|
||||||
pub fn powerInformation() PowerInformation {
|
pub fn powerInformation() PowerInformation {
|
||||||
return acpi.power_information;
|
return acpi.power_information;
|
||||||
}
|
}
|
||||||
@@ -39,18 +40,6 @@ pub fn platformInformation() PlatformInformation {
|
|||||||
return acpi.platform_information;
|
return acpi.platform_information;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
|
||||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
|
||||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
|
||||||
/// count against this.
|
|
||||||
pub fn amlDeviceCount() usize {
|
|
||||||
return acpi.amlDeviceCount();
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn amlStats() AmlStats {
|
|
||||||
return acpi.aml_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The usable logical processors discovered during enumeration — one entry per
|
/// The usable logical processors discovered during enumeration — one entry per
|
||||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||||
@@ -92,12 +81,8 @@ pub fn discover(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls. Soft-off
|
||||||
|
/// (S5) is not a kernel operation — the ring-3 acpi service owns it (docs/power.md).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
power.reboot(hal);
|
power.reboot(hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Power the machine off (ACPI S5). Never returns on success.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
power.shutdown(hal);
|
|
||||||
}
|
|
||||||
|
|||||||
+10
-63
@@ -1,34 +1,17 @@
|
|||||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
//! Machine reboot: restart via the FADT reset register, with legacy fallbacks.
|
||||||
//!
|
//!
|
||||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
//! Built on the register map `acpi` extracted from the FADT, driven through the
|
||||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
//! injected `Hal` (port I/O and MMIO). Soft-off (ACPI S5) and suspend (S3) are
|
||||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
//! **not** here: they need the AML sleep-state (`_Sx`) values, which the kernel no
|
||||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
//! longer parses — the ring-3 acpi service owns power management (it re-parses the
|
||||||
//!
|
//! blobs and writes the PM1 control register itself). See docs/power.md. Reboot
|
||||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
//! stays in the kernel because it needs no AML — only the FADT reset register and
|
||||||
//! re-initialisation, a milestone of its own.
|
//! the well-known legacy fallbacks — so it survives as a last-resort restart.
|
||||||
|
|
||||||
const acpi = @import("acpi.zig");
|
const acpi = @import("acpi.zig");
|
||||||
const device_model = @import("device-model.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const Hal = device_model.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
|
||||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
|
||||||
|
|
||||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
|
||||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
|
||||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
|
||||||
pub fn enable(hal: Hal) void {
|
|
||||||
const pi = acpi.power_information;
|
|
||||||
if (!pi.pm1a_cnt.present()) return;
|
|
||||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
|
||||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
|
||||||
|
|
||||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
|
||||||
var spins: usize = 0;
|
|
||||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
@@ -48,42 +31,6 @@ pub fn reboot(hal: Hal) void {
|
|||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
|
||||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
enable(hal);
|
|
||||||
const pi = acpi.power_information;
|
|
||||||
const s5 = pi.s5 orelse return;
|
|
||||||
|
|
||||||
if (pi.pm1a_cnt.present()) {
|
|
||||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
|
||||||
}
|
|
||||||
if (pi.pm1b_cnt.present()) {
|
|
||||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
|
||||||
}
|
|
||||||
delay();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
|
||||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
|
||||||
_ = hal;
|
|
||||||
return error.Unsupported;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
|
||||||
/// [12:10], SLP_EN in bit 13.
|
|
||||||
fn sleepValue(slp_typ: u8) u32 {
|
|
||||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
|
||||||
if (register.mmio) {
|
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
|
||||||
return p.*;
|
|
||||||
}
|
|
||||||
return hal.pioRead(register.width, @intCast(register.address));
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||||
if (register.mmio) {
|
if (register.mmio) {
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
@@ -93,8 +40,8 @@ fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
/// A short busy-wait so a reset takes effect before we fall through to the next
|
||||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
/// method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||||
/// from being optimised away.
|
/// from being optimised away.
|
||||||
fn delay() void {
|
fn delay() void {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ const op_inquiry: u8 = 0x12;
|
|||||||
const op_read_capacity_10: u8 = 0x25;
|
const op_read_capacity_10: u8 = 0x25;
|
||||||
const op_read_10: u8 = 0x28;
|
const op_read_10: u8 = 0x28;
|
||||||
const op_write_10: u8 = 0x2A;
|
const op_write_10: u8 = 0x2A;
|
||||||
|
const op_synchronize_cache_10: u8 = 0x35;
|
||||||
|
|
||||||
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
||||||
/// and product strings).
|
/// and product strings).
|
||||||
@@ -57,6 +58,16 @@ pub fn write10(lba: u32, blocks: u16) [10]u8 {
|
|||||||
return cdb;
|
return cdb;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// SYNCHRONIZE CACHE(10): commit the device's write cache to stable media. LBA 0
|
||||||
|
/// and block count 0 mean "the whole medium". No data stage. Without this a write
|
||||||
|
/// can sit in the USB flash controller's cache and be lost if power is cut right
|
||||||
|
/// after — which is exactly what a shutdown-time log flush hits on real hardware.
|
||||||
|
pub fn synchronizeCache10() [10]u8 {
|
||||||
|
var cdb = [_]u8{0} ** 10;
|
||||||
|
cdb[0] = op_synchronize_cache_10;
|
||||||
|
return cdb;
|
||||||
|
}
|
||||||
|
|
||||||
/// Decode an 8-byte READ CAPACITY(10) reply.
|
/// Decode an 8-byte READ CAPACITY(10) reply.
|
||||||
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
||||||
return .{
|
return .{
|
||||||
|
|||||||
@@ -147,6 +147,14 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
|||||||
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||||
},
|
},
|
||||||
|
@intFromEnum(block_protocol.Operation.flush) => {
|
||||||
|
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||||
|
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||||
|
// cuts power. A device without a volatile cache reports success anyway.
|
||||||
|
const cdb = scsi.synchronizeCache10();
|
||||||
|
const ok = transact(&cdb, false, 0, 0);
|
||||||
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||||
|
},
|
||||||
else => return 0,
|
else => return 0,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,139 @@
|
|||||||
|
//! The virtio-gpu control protocol — the command/response structs the driver exchanges with
|
||||||
|
//! the device over its control virtqueue (virtio spec, "GPU Device"). `extern` structs, so
|
||||||
|
//! the layout matches the little-endian wire format exactly. Host-tested for size. See
|
||||||
|
//! docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Control command / response types (virtio_gpu_ctrl_type). Commands are 0x01xx, responses
|
||||||
|
/// 0x11xx (ok) / 0x12xx (error).
|
||||||
|
pub const CmdType = enum(u32) {
|
||||||
|
get_display_info = 0x0100,
|
||||||
|
resource_create_2d = 0x0101,
|
||||||
|
resource_unref = 0x0102,
|
||||||
|
set_scanout = 0x0103,
|
||||||
|
resource_flush = 0x0104,
|
||||||
|
transfer_to_host_2d = 0x0105,
|
||||||
|
resource_attach_backing = 0x0106,
|
||||||
|
resource_detach_backing = 0x0107,
|
||||||
|
get_edid = 0x010a,
|
||||||
|
|
||||||
|
resp_ok_nodata = 0x1100,
|
||||||
|
resp_ok_display_info = 0x1101,
|
||||||
|
resp_ok_edid = 0x1104,
|
||||||
|
resp_err_unspec = 0x1200,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Set in a command's `flags` to request a fence; the device echoes `fence_id` in the
|
||||||
|
/// response and does not report completion until the command's effects are visible.
|
||||||
|
pub const flag_fence: u32 = 1 << 0;
|
||||||
|
|
||||||
|
/// VIRTIO_GPU_F_EDID — device feature bit 1 (the low feature word): the device answers the
|
||||||
|
/// `get_edid` command. Negotiate it only when the device offers it.
|
||||||
|
pub const feature_edid: u32 = 1 << 1;
|
||||||
|
|
||||||
|
/// virtio_gpu_ctrl_hdr — the header on every command and response.
|
||||||
|
pub const CtrlHdr = extern struct {
|
||||||
|
type: u32,
|
||||||
|
flags: u32 = 0,
|
||||||
|
fence_id: u64 = 0,
|
||||||
|
ctx_id: u32 = 0,
|
||||||
|
ring_idx: u8 = 0,
|
||||||
|
padding: [3]u8 = .{ 0, 0, 0 },
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Rect = extern struct {
|
||||||
|
x: u32,
|
||||||
|
y: u32,
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 2D pixel formats. QEMU's virtio-gpu host default is B8G8R8X8 (matches our bgrx).
|
||||||
|
pub const format_b8g8r8x8_unorm: u32 = 2;
|
||||||
|
pub const format_r8g8b8x8_unorm: u32 = 134;
|
||||||
|
|
||||||
|
pub const ResourceCreate2d = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
resource_id: u32,
|
||||||
|
format: u32,
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One scatter-gather entry of a resource's guest backing (a physical span).
|
||||||
|
pub const MemEntry = extern struct {
|
||||||
|
addr: u64,
|
||||||
|
length: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Header for RESOURCE_ATTACH_BACKING; `nr_entries` `MemEntry` follow it inline.
|
||||||
|
pub const ResourceAttachBacking = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
resource_id: u32,
|
||||||
|
nr_entries: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const SetScanout = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
rect: Rect,
|
||||||
|
scanout_id: u32,
|
||||||
|
resource_id: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const ResourceFlush = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
rect: Rect,
|
||||||
|
resource_id: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Copy the guest backing into the host resource for `rect` (2D resources must transfer
|
||||||
|
/// before a flush shows the update).
|
||||||
|
pub const TransferToHost2d = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
rect: Rect,
|
||||||
|
offset: u64,
|
||||||
|
resource_id: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const max_scanouts = 16;
|
||||||
|
|
||||||
|
pub const DisplayOne = extern struct {
|
||||||
|
rect: Rect,
|
||||||
|
enabled: u32,
|
||||||
|
flags: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const RespDisplayInfo = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
pmodes: [max_scanouts]DisplayOne,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const GetEdid = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
scanout: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const RespEdid = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
size: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
edid: [1024]u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
test "virtio-gpu struct sizes match the wire layout" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 24), @sizeOf(CtrlHdr));
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Rect));
|
||||||
|
try std.testing.expectEqual(@as(usize, 40), @sizeOf(ResourceCreate2d));
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(MemEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(ResourceAttachBacking));
|
||||||
|
try std.testing.expectEqual(@as(usize, 48), @sizeOf(SetScanout));
|
||||||
|
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ResourceFlush));
|
||||||
|
try std.testing.expectEqual(@as(usize, 56), @sizeOf(TransferToHost2d));
|
||||||
|
try std.testing.expectEqual(@as(usize, 24 + 4 + 4 + 1024), @sizeOf(RespEdid));
|
||||||
|
}
|
||||||
@@ -0,0 +1,622 @@
|
|||||||
|
//! /system/drivers/virtio-gpu — the virtio-gpu (virtio 1.0, modern PCI) display driver.
|
||||||
|
//! The device manager spawns it for the display/other PCI function (class 0x0380) whose
|
||||||
|
//! config space says vendor 0x1AF4 / device 0x1050; this instance claims that device and
|
||||||
|
//! brings up a single 2D scanout.
|
||||||
|
//!
|
||||||
|
//! V3 (this increment): the whole path end to end, proven from serial without a screenshot.
|
||||||
|
//! Claim the function, map its config space (resource 0) and the BAR that carries the
|
||||||
|
//! virtio structures, walk the vendor capabilities to find common-config / notify, reset
|
||||||
|
//! and negotiate VERSION_1, stand up the control virtqueue in DMA memory, then drive the
|
||||||
|
//! GPU: RESOURCE_CREATE_2D → ATTACH_BACKING (a coherent DMA buffer) → SET_SCANOUT, paint a
|
||||||
|
//! known test pattern, TRANSFER_TO_HOST_2D → RESOURCE_FLUSH, and **wait for the device's
|
||||||
|
//! used-ring ack**. Reading the backing back confirms it is CPU-visible; the ack confirms
|
||||||
|
//! the device consumed the frame. The compositor backend, hot-attach, mode-set/EDID, and
|
||||||
|
//! restart/re-attach are V4–V6. See docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const mmio = @import("mmio");
|
||||||
|
const device = runtime.device;
|
||||||
|
const dma = runtime.dma;
|
||||||
|
const shm = runtime.shm;
|
||||||
|
const system = runtime.system;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const dp = runtime.display_protocol;
|
||||||
|
const sp = runtime.scanout_protocol;
|
||||||
|
const dm = runtime.device_manager_protocol;
|
||||||
|
const vp = @import("virtio-pci.zig");
|
||||||
|
const vg = @import("virtio-gpu-protocol.zig");
|
||||||
|
|
||||||
|
/// The DisplayFormat (device-abi) our B8G8R8X8 scanout resource presents: bgrx = 1. Handed to
|
||||||
|
/// the compositor in the announce so it packs colours in the surface's byte order.
|
||||||
|
const display_format_bgrx: u32 = 1;
|
||||||
|
|
||||||
|
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
|
||||||
|
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
|
||||||
|
const virtio_vendor: u16 = 0x1AF4;
|
||||||
|
const virtio_gpu_device: u16 = 0x1050;
|
||||||
|
|
||||||
|
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
||||||
|
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
||||||
|
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
||||||
|
/// the compositor is told in the announce. Kept modest so the backing is an easy contiguous run.
|
||||||
|
const max_width: u32 = 800;
|
||||||
|
const max_height: u32 = 600;
|
||||||
|
const scanout_bytes: usize = @as(usize, max_width) * max_height * 4;
|
||||||
|
const resource_id: u32 = 1;
|
||||||
|
|
||||||
|
/// The modes this scanout offers (all ≤ max). The first is the mode it comes up in.
|
||||||
|
const Mode = struct { width: u32, height: u32 };
|
||||||
|
const offered_modes = [_]Mode{ .{ .width = 640, .height = 480 }, .{ .width = 800, .height = 600 } };
|
||||||
|
|
||||||
|
/// The active mode — the scanout rectangle within the max-sized surface. Changed by `set_mode`.
|
||||||
|
var current_width: u32 = offered_modes[0].width;
|
||||||
|
var current_height: u32 = offered_modes[0].height;
|
||||||
|
|
||||||
|
/// Monotonic fence id for fenced (vsync) flushes; the device signals the fence when the flush
|
||||||
|
/// is complete, which its used-ring ack already gates our synchronous present on.
|
||||||
|
var fence_next: u64 = 1;
|
||||||
|
|
||||||
|
/// Whether the device offered VIRTIO_GPU_F_EDID, so `get_edid` is worth issuing.
|
||||||
|
var edid_available = false;
|
||||||
|
|
||||||
|
/// The control virtqueue. We drive it synchronously — one command, notify, poll the used
|
||||||
|
/// ring — so a depth of 16 is ample; we ask the device to shrink to it (virtio 1.0 lets the
|
||||||
|
/// driver reduce queue_size), keeping the whole ring inside one page.
|
||||||
|
const queue_size: u16 = 16;
|
||||||
|
const desc_offset: usize = 0; // 16 * 16 = 256 bytes
|
||||||
|
const avail_offset: usize = 256; // flags + idx + ring[16] + used_event = 38 bytes
|
||||||
|
const used_offset: usize = 1024; // flags + idx + ring[16] + avail_event = 134 bytes
|
||||||
|
|
||||||
|
/// The command scratch: the request the device reads, then its response, in one DMA page.
|
||||||
|
const request_offset: usize = 0;
|
||||||
|
const response_offset: usize = 2048;
|
||||||
|
|
||||||
|
var device_id: u64 = 0;
|
||||||
|
|
||||||
|
// Mapped virtio structures (virtual addresses into the device's BAR).
|
||||||
|
var common_base: usize = 0;
|
||||||
|
var notify_base: usize = 0;
|
||||||
|
var notify_multiplier: u32 = 0;
|
||||||
|
var notify_addr: usize = 0;
|
||||||
|
|
||||||
|
// Per-BAR mapping cache: several capabilities usually share one BAR, and mmio_map must not
|
||||||
|
// be asked to map the same resource twice.
|
||||||
|
var bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 };
|
||||||
|
|
||||||
|
// DMA memory: the virtqueue rings and the command scratch.
|
||||||
|
var ring: dma.Region = undefined;
|
||||||
|
var command: dma.Region = undefined;
|
||||||
|
|
||||||
|
// The scanout backing is a **shared** (shm) region, not DMA: cacheable so the compositor
|
||||||
|
// composites into it cheaply (x86 DMA is coherent, so the device still sees the writes), and
|
||||||
|
// shareable so the same physical pages the device scans out of are the ones the compositor
|
||||||
|
// paints. The driver keeps the capability to hand to the compositor in the announce.
|
||||||
|
var surface: shm.Region = undefined;
|
||||||
|
|
||||||
|
// Split-virtqueue producer/consumer shadows.
|
||||||
|
var avail_shadow: u16 = 0;
|
||||||
|
var used_shadow: u16 = 0;
|
||||||
|
|
||||||
|
/// Format one whole log line and emit it in a single `write`, so this driver's output can
|
||||||
|
/// never interleave mid-line with the other drivers the manager runs concurrently.
|
||||||
|
fn log(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [160]u8 = undefined;
|
||||||
|
_ = system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- common-config register access (little-endian MMIO at `common_base`) ---------------
|
||||||
|
|
||||||
|
fn cfgRead(comptime T: type, comptime field: []const u8) T {
|
||||||
|
return mmio.read(T, common_base + @offsetOf(vp.CommonCfg, field));
|
||||||
|
}
|
||||||
|
fn cfgWrite(comptime T: type, comptime field: []const u8, value: T) void {
|
||||||
|
mmio.write(T, common_base + @offsetOf(vp.CommonCfg, field), value);
|
||||||
|
}
|
||||||
|
/// Write a 64-bit common-config register as two 32-bit halves (low then high) — the widest
|
||||||
|
/// access every virtio-pci host is required to accept for the queue-address registers.
|
||||||
|
fn cfgWrite64(comptime field: []const u8, value: u64) void {
|
||||||
|
const at = common_base + @offsetOf(vp.CommonCfg, field);
|
||||||
|
mmio.write(u32, at, @truncate(value));
|
||||||
|
mmio.write(u32, at + 4, @truncate(value >> 32));
|
||||||
|
}
|
||||||
|
fn orStatus(bit: u8) void {
|
||||||
|
cfgWrite(u8, "device_status", cfgRead(u8, "device_status") | bit);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- PCI config-space capability walk (config space is resource 0) ---------------------
|
||||||
|
|
||||||
|
/// Map the BAR numbered `bar` (0..5) and return its virtual base, correlating the BAR's
|
||||||
|
/// physical address (read from config space) with one of our device resources — because a
|
||||||
|
/// virtio capability names a BAR *number*, while `mmio_map` takes a *resource index* (and
|
||||||
|
/// resource 0 is config space, so BAR resources are re-numbered and gaps skipped).
|
||||||
|
fn mapBar(config: usize, descriptor: *const device.DeviceDescriptor, bar: u8) ?usize {
|
||||||
|
if (bar >= 6) return null;
|
||||||
|
if (bar_virtual[bar] != 0) return bar_virtual[bar];
|
||||||
|
|
||||||
|
const low = mmio.read(u32, config + 0x10 + @as(usize, bar) * 4);
|
||||||
|
if (low & 0x1 != 0) return null; // an I/O-space BAR — virtio structures are in memory BARs
|
||||||
|
var base: u64 = low & 0xFFFF_FFF0;
|
||||||
|
if ((low & 0x6) == 0x4) { // 64-bit memory BAR: the high half is the next dword
|
||||||
|
const high = mmio.read(u32, config + 0x10 + (@as(usize, bar) + 1) * 4);
|
||||||
|
base |= @as(u64, high) << 32;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||||
|
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||||
|
const v = device.mmioMap(device_id, index) orelse return null;
|
||||||
|
bar_virtual[bar] = v;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
log("virtio-gpu: BAR {d} (physical 0x{x}) is not a mapped resource\n", .{ bar, base });
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Walk the PCI capability list from mapped config space, recording the common-config and
|
||||||
|
/// notify structures (the only two V3 needs). Returns false if either is missing.
|
||||||
|
fn walkCapabilities(config: usize, descriptor: *const device.DeviceDescriptor) bool {
|
||||||
|
if (mmio.read(u16, config + 0x06) & 0x10 == 0) { // Status bit 4: capabilities list present
|
||||||
|
log("virtio-gpu: device has no PCI capability list\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
var cap: u8 = @as(u8, @truncate(mmio.read(u8, config + 0x34))) & 0xFC;
|
||||||
|
var guard: u32 = 0;
|
||||||
|
while (cap != 0 and guard < 48) : (guard += 1) {
|
||||||
|
const at = config + cap;
|
||||||
|
const id = mmio.read(u8, at + 0);
|
||||||
|
const next = mmio.read(u8, at + 1) & 0xFC;
|
||||||
|
// Only map BARs for the structures V3 uses (common + notify). The other virtio
|
||||||
|
// capabilities (isr, device, and especially the cfg_pci back-door, which carries a
|
||||||
|
// placeholder bar=0/offset=0) reference BARs we never touch, so mapping them would
|
||||||
|
// just log spurious "not a mapped resource" noise.
|
||||||
|
if (id == vp.pci_cap_vendor) {
|
||||||
|
const cfg_type = mmio.read(u8, at + 3);
|
||||||
|
if (cfg_type == vp.cfg_common or cfg_type == vp.cfg_notify) {
|
||||||
|
const bar = mmio.read(u8, at + 4);
|
||||||
|
const offset = mmio.read(u32, at + 8);
|
||||||
|
if (mapBar(config, descriptor, bar)) |bar_base| {
|
||||||
|
if (cfg_type == vp.cfg_common) {
|
||||||
|
common_base = bar_base + offset;
|
||||||
|
} else {
|
||||||
|
notify_base = bar_base + offset;
|
||||||
|
notify_multiplier = mmio.read(u32, at + 16); // virtio_pci_notify_cap tail
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
cap = next;
|
||||||
|
}
|
||||||
|
if (common_base == 0 or notify_base == 0) {
|
||||||
|
log("virtio-gpu: missing common-config or notify capability\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the control virtqueue -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Publish the two-descriptor chain (request read by the device, response written by it),
|
||||||
|
/// notify the control queue, and wait for the device to return the buffer on the used ring.
|
||||||
|
fn submit(request_len: usize, response_len: usize) bool {
|
||||||
|
const desc: [*]vp.Desc = @ptrFromInt(ring.virtual + desc_offset);
|
||||||
|
desc[0] = .{
|
||||||
|
.addr = command.physical + request_offset,
|
||||||
|
.len = @intCast(request_len),
|
||||||
|
.flags = vp.desc_flag_next,
|
||||||
|
.next = 1,
|
||||||
|
};
|
||||||
|
desc[1] = .{
|
||||||
|
.addr = command.physical + response_offset,
|
||||||
|
.len = @intCast(response_len),
|
||||||
|
.flags = vp.desc_flag_write,
|
||||||
|
.next = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
const avail_ring: [*]u16 = @ptrFromInt(ring.virtual + avail_offset + 4);
|
||||||
|
avail_ring[avail_shadow % queue_size] = 0; // head of the chain is descriptor 0
|
||||||
|
mmio.wmb();
|
||||||
|
avail_shadow +%= 1;
|
||||||
|
mmio.write(u16, ring.virtual + avail_offset + 2, avail_shadow); // avail.idx
|
||||||
|
mmio.wmb();
|
||||||
|
|
||||||
|
mmio.write(u16, notify_addr, 0); // ring the control queue's doorbell
|
||||||
|
return waitUsed();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spin, then sleep-poll, on the used-ring index until the device advances it. QEMU
|
||||||
|
/// processes the notify on its own thread, so the ack usually lands immediately; the sleep
|
||||||
|
/// fallback covers a device that defers it without burning the CPU.
|
||||||
|
fn waitUsed() bool {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (tries < 2000) : (tries += 1) {
|
||||||
|
mmio.rmb();
|
||||||
|
const idx = mmio.read(u16, ring.virtual + used_offset + 2); // used.idx
|
||||||
|
if (idx != used_shadow) {
|
||||||
|
used_shadow = idx;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (tries > 8) system.sleep(1);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The type field of the response the device wrote — `resp_ok_nodata` on success.
|
||||||
|
fn responseType() u32 {
|
||||||
|
const response: *vg.CtrlHdr = @ptrFromInt(command.virtual + response_offset);
|
||||||
|
return response.type;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Submit a command whose response is a bare header, returning its response type (0 if the
|
||||||
|
/// device never acked).
|
||||||
|
fn command_nodata(request_len: usize) u32 {
|
||||||
|
if (!submit(request_len, @sizeOf(vg.CtrlHdr))) return 0;
|
||||||
|
return responseType();
|
||||||
|
}
|
||||||
|
|
||||||
|
const ok_nodata: u32 = @intFromEnum(vg.CmdType.resp_ok_nodata);
|
||||||
|
|
||||||
|
fn requestAt(comptime T: type) *T {
|
||||||
|
return @ptrFromInt(command.virtual + request_offset);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A deterministic, recognisable pixel so a read-back is a real check, not a tautology.
|
||||||
|
fn testPixel(index: u32) u32 {
|
||||||
|
return 0xFF00_0000 | (index *% 0x9E37_79B1);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- bring-up --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
|
_ = endpoint;
|
||||||
|
if (!device.claim(device_id)) {
|
||||||
|
log("virtio-gpu: unable to claim device {d}\n", .{device_id});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||||
|
const total = device.enumerate(&descriptors);
|
||||||
|
const descriptor = for (descriptors[0..@min(total, descriptors.len)]) |*d| {
|
||||||
|
if (d.id == device_id) break d;
|
||||||
|
} else {
|
||||||
|
log("virtio-gpu: device {d} not in the device tree\n", .{device_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Config space is resource 0. Confirm it really is a virtio-gpu, then enable memory-space
|
||||||
|
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
||||||
|
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
||||||
|
const config = device.mmioMap(device_id, 0) orelse {
|
||||||
|
log("virtio-gpu: config-space map failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const vendor = mmio.read(u16, config + 0x00);
|
||||||
|
const dev = mmio.read(u16, config + 0x02);
|
||||||
|
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
||||||
|
log("virtio-gpu: not a virtio-gpu (vendor 0x{x} device 0x{x})\n", .{ vendor, dev });
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
mmio.write(u16, config + 0x04, mmio.read(u16, config + 0x04) | 0x06); // MEM + bus master
|
||||||
|
|
||||||
|
if (!walkCapabilities(config, descriptor)) return false;
|
||||||
|
|
||||||
|
// Reset, then the modern feature handshake: acknowledge, take driver ownership, require
|
||||||
|
// VERSION_1 and offer nothing else, and confirm the device accepts that.
|
||||||
|
cfgWrite(u8, "device_status", 0);
|
||||||
|
orStatus(vp.status_acknowledge);
|
||||||
|
orStatus(vp.status_driver);
|
||||||
|
|
||||||
|
// Low feature word (device-specific): note whether the device offers EDID (bit 1).
|
||||||
|
cfgWrite(u32, "device_feature_select", 0);
|
||||||
|
edid_available = cfgRead(u32, "device_feature") & vg.feature_edid != 0;
|
||||||
|
// High feature word: VERSION_1 (bit 32) is required for a modern device.
|
||||||
|
cfgWrite(u32, "device_feature_select", vp.feature_version_1_word);
|
||||||
|
if (cfgRead(u32, "device_feature") & vp.feature_version_1_bit == 0) {
|
||||||
|
log("virtio-gpu: device does not offer VERSION_1 (not a modern device)\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// Accept exactly VERSION_1, plus EDID when the device offered it (never a feature it didn't).
|
||||||
|
cfgWrite(u32, "driver_feature_select", 0);
|
||||||
|
cfgWrite(u32, "driver_feature", if (edid_available) vg.feature_edid else 0);
|
||||||
|
cfgWrite(u32, "driver_feature_select", vp.feature_version_1_word);
|
||||||
|
cfgWrite(u32, "driver_feature", vp.feature_version_1_bit);
|
||||||
|
orStatus(vp.status_features_ok);
|
||||||
|
if (cfgRead(u8, "device_status") & vp.status_features_ok == 0) {
|
||||||
|
log("virtio-gpu: device rejected the negotiated features\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Stand up the control virtqueue (queue 0) in coherent DMA memory.
|
||||||
|
cfgWrite(u16, "queue_select", 0);
|
||||||
|
const device_qsize = cfgRead(u16, "queue_size");
|
||||||
|
if (device_qsize < queue_size) {
|
||||||
|
log("virtio-gpu: control queue too small ({d})\n", .{device_qsize});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
ring = dma.alloc(4096, dma.coherent) orelse {
|
||||||
|
log("virtio-gpu: virtqueue allocation failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
command = dma.alloc(4096, dma.coherent) orelse {
|
||||||
|
log("virtio-gpu: command-buffer allocation failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
mmio.write(u16, ring.virtual + avail_offset, 1); // VIRTQ_AVAIL_F_NO_INTERRUPT: we poll
|
||||||
|
cfgWrite(u16, "queue_size", queue_size);
|
||||||
|
cfgWrite64("queue_desc", ring.physical + desc_offset);
|
||||||
|
cfgWrite64("queue_driver", ring.physical + avail_offset);
|
||||||
|
cfgWrite64("queue_device", ring.physical + used_offset);
|
||||||
|
cfgWrite(u16, "queue_msix_vector", 0xFFFF); // VIRTIO_MSI_NO_VECTOR
|
||||||
|
cfgWrite(u16, "queue_enable", 1);
|
||||||
|
|
||||||
|
cfgWrite(u16, "queue_select", 0);
|
||||||
|
notify_addr = notify_base + @as(usize, cfgRead(u16, "queue_notify_off")) * notify_multiplier;
|
||||||
|
|
||||||
|
orStatus(vp.status_driver_ok);
|
||||||
|
|
||||||
|
// Drive the GPU: create a 2D resource at the *max* mode, back it with a shared surface, and
|
||||||
|
// scan out the current-mode rectangle within it.
|
||||||
|
{
|
||||||
|
const request = requestAt(vg.ResourceCreate2d);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_create_2d) },
|
||||||
|
.resource_id = resource_id,
|
||||||
|
.format = vg.format_b8g8r8x8_unorm,
|
||||||
|
.width = max_width,
|
||||||
|
.height = max_height,
|
||||||
|
};
|
||||||
|
if (command_nodata(@sizeOf(vg.ResourceCreate2d)) != ok_nodata) {
|
||||||
|
log("virtio-gpu: resource_create_2d failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Back the resource with a shared (shm) surface, so the compositor and the device work
|
||||||
|
// the same physical pages. The device needs the guest-physical base for attach_backing.
|
||||||
|
surface = shm.create(scanout_bytes) orelse {
|
||||||
|
log("virtio-gpu: scanout surface allocation failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const surface_physical = shm.physical(surface.handle) orelse {
|
||||||
|
log("virtio-gpu: could not resolve the scanout surface physical address\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
{
|
||||||
|
const request = requestAt(vg.ResourceAttachBacking);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_attach_backing) },
|
||||||
|
.resource_id = resource_id,
|
||||||
|
.nr_entries = 1,
|
||||||
|
};
|
||||||
|
const entry: *vg.MemEntry = @ptrFromInt(command.virtual + request_offset + @sizeOf(vg.ResourceAttachBacking));
|
||||||
|
entry.* = .{ .addr = surface_physical, .length = @intCast(scanout_bytes) };
|
||||||
|
if (command_nodata(@sizeOf(vg.ResourceAttachBacking) + @sizeOf(vg.MemEntry)) != ok_nodata) {
|
||||||
|
log("virtio-gpu: resource_attach_backing failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!setScanoutRect()) {
|
||||||
|
log("virtio-gpu: set_scanout failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
log("virtio-gpu: scanout {d}x{d} online\n", .{ current_width, current_height });
|
||||||
|
|
||||||
|
// Hello the device manager so it counts us as up (and does not stop us at the hello
|
||||||
|
// deadline). A restarted instance re-hellos here and re-announces below — the compositor
|
||||||
|
// re-attaches to the fresh scanout (V6).
|
||||||
|
helloManager();
|
||||||
|
|
||||||
|
// Read the monitor's EDID (best-effort, when the device offers it) — the mode list a real
|
||||||
|
// driver derives from it; we log the preferred mode and keep our fixed offered list.
|
||||||
|
readEdid();
|
||||||
|
|
||||||
|
// Paint a known pattern, present it, and read it back — the V3 self-test that proves the
|
||||||
|
// whole path (virtqueue, resource, shared backing, transfer, flush) before a client attaches.
|
||||||
|
const pixels: [*]u32 = @ptrCast(@alignCast(surface.ptr));
|
||||||
|
const pixel_count: usize = @as(usize, max_width) * max_height;
|
||||||
|
for (0..pixel_count) |i| pixels[i] = testPixel(@intCast(i));
|
||||||
|
|
||||||
|
if (!presentFull()) {
|
||||||
|
log("virtio-gpu: initial present failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// The scanout surface is CPU-visible RAM: read the pattern back to prove the mapping,
|
||||||
|
// which together with the flush ack above is the automated stand-in for "it's on screen".
|
||||||
|
mmio.rmb();
|
||||||
|
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(@intCast(pixel_count / 2))) {
|
||||||
|
log("virtio-gpu: pixel read-back mismatch\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
log("virtio-gpu: flush acked, pixel check ok\n", .{});
|
||||||
|
|
||||||
|
// Offer the shared surface to the compositor so it upgrades off the GOP floor (V4).
|
||||||
|
announce();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Point scanout 0 at the current-mode rectangle of the resource. Reused by initial bring-up
|
||||||
|
/// and by `set_mode`.
|
||||||
|
fn setScanoutRect() bool {
|
||||||
|
const request = requestAt(vg.SetScanout);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.set_scanout) },
|
||||||
|
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||||
|
.scanout_id = 0,
|
||||||
|
.resource_id = resource_id,
|
||||||
|
};
|
||||||
|
return command_nodata(@sizeOf(vg.SetScanout)) == ok_nodata;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read and log the monitor's preferred mode from its EDID (VIRTIO_GPU_F_EDID). Best-effort:
|
||||||
|
/// a device that doesn't offer EDID, or a missing/short block, is logged and ignored.
|
||||||
|
fn readEdid() void {
|
||||||
|
if (!edid_available) {
|
||||||
|
log("virtio-gpu: EDID not offered by device\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const request = requestAt(vg.GetEdid);
|
||||||
|
request.* = .{ .hdr = .{ .type = @intFromEnum(vg.CmdType.get_edid) }, .scanout = 0 };
|
||||||
|
if (!submit(@sizeOf(vg.GetEdid), @sizeOf(vg.RespEdid))) {
|
||||||
|
log("virtio-gpu: EDID request not acked\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const response: *vg.RespEdid = @ptrFromInt(command.virtual + response_offset);
|
||||||
|
if (response.hdr.type != @intFromEnum(vg.CmdType.resp_ok_edid) or response.size < 64) {
|
||||||
|
log("virtio-gpu: EDID unavailable\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// The first detailed timing descriptor (EDID base-block offset 54) is the preferred mode:
|
||||||
|
// active pixels are 12-bit, low byte + high nibble (bytes 2/4 horizontal, 5/7 vertical).
|
||||||
|
const e = &response.edid;
|
||||||
|
const h_active = @as(u32, e[56]) | (@as(u32, e[58] & 0xF0) << 4);
|
||||||
|
const v_active = @as(u32, e[59]) | (@as(u32, e[61] & 0xF0) << 4);
|
||||||
|
log("virtio-gpu: EDID preferred mode {d}x{d}\n", .{ h_active, v_active });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Present the whole surface: copy the guest backing into the host resource, then flush it to
|
||||||
|
/// the panel. Reused by the V3 self-test and by every compositor present over `.scanout`. V4
|
||||||
|
/// presents the full surface; the damage-rect fast path is a later refinement.
|
||||||
|
fn presentFull() bool {
|
||||||
|
mmio.wmb(); // the surface writes must be visible before the device transfers them
|
||||||
|
{
|
||||||
|
// Transfer the current-mode rectangle from the guest backing to the host resource. The
|
||||||
|
// device uses the resource's (max) width as the row stride, so the top-left rect at
|
||||||
|
// offset 0 is exactly the visible area — the compositor composes at that same stride.
|
||||||
|
const request = requestAt(vg.TransferToHost2d);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.transfer_to_host_2d) },
|
||||||
|
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||||
|
.offset = 0,
|
||||||
|
.resource_id = resource_id,
|
||||||
|
};
|
||||||
|
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) return false;
|
||||||
|
}
|
||||||
|
{
|
||||||
|
// A fenced flush (vsync): the device signals the fence when the frame is actually on
|
||||||
|
// screen — which its used-ring ack, what our synchronous submit waits on, already gates.
|
||||||
|
const request = requestAt(vg.ResourceFlush);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_flush), .flags = vg.flag_fence, .fence_id = fence_next },
|
||||||
|
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||||
|
.resource_id = resource_id,
|
||||||
|
};
|
||||||
|
fence_next += 1;
|
||||||
|
if (command_nodata(@sizeOf(vg.ResourceFlush)) != ok_nodata) return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hello the device manager (role: bus — we own a PCI function, though we report no children):
|
||||||
|
/// the handshake that marks us up so the manager doesn't stop us at the hello deadline, and
|
||||||
|
/// (as a supervised driver) restarts us if we die. Best-effort: without a manager we still run.
|
||||||
|
fn helloManager() void {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
const manager = while (tries < 100) : (tries += 1) {
|
||||||
|
if (ipc.lookup(.device_manager)) |h| break h;
|
||||||
|
system.sleep(20);
|
||||||
|
} else {
|
||||||
|
log("virtio-gpu: no device manager to hello\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const hello = dm.Hello{ .role = @intFromEnum(dm.Role.bus), .device_id = device_id };
|
||||||
|
var reply: [dm.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch {
|
||||||
|
log("virtio-gpu: hello call failed\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (n < dm.reply_size or std.mem.bytesToValue(dm.HelloReply, reply[0..dm.reply_size]).status != 0) {
|
||||||
|
log("virtio-gpu: hello refused\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
log("virtio-gpu: hello acknowledged\n", .{});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Announce the scanout to the display service so it upgrades off the GOP framebuffer: hand it
|
||||||
|
/// the shared surface as a capability plus the geometry. Best-effort and non-fatal — without a
|
||||||
|
/// display service (the standalone virtio-gpu bring-up test) the driver is still a valid
|
||||||
|
/// scanout service; it just serves no one. The display replies immediately (it defers its
|
||||||
|
/// first present to a timer), so this returns before we start serving `.scanout` — no deadlock.
|
||||||
|
fn announce() void {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
const display = while (tries < 50) : (tries += 1) {
|
||||||
|
if (ipc.lookup(.display)) |h| break h;
|
||||||
|
system.sleep(20);
|
||||||
|
} else {
|
||||||
|
log("virtio-gpu: no display service to announce to (scanout-only)\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var request = dp.Request{
|
||||||
|
.operation = @intFromEnum(dp.Operation.attach_scanout),
|
||||||
|
.x = max_width, // the shared surface's row stride in pixels (it is sized to the max mode)
|
||||||
|
.width = current_width,
|
||||||
|
.height = current_height,
|
||||||
|
.colour = display_format_bgrx,
|
||||||
|
};
|
||||||
|
var reply: [dp.reply_size]u8 = undefined;
|
||||||
|
_ = ipc.callCap(display, std.mem.asBytes(&request), &reply, surface.handle) catch {
|
||||||
|
log("virtio-gpu: announce to display failed\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
log("virtio-gpu: announced scanout to display\n", .{});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A `sp.Reply{status}` written into `reply`.
|
||||||
|
fn scanoutStatus(reply: []u8, ok: bool) usize {
|
||||||
|
const response = sp.Reply{ .status = if (ok) 0 else -1 };
|
||||||
|
@memcpy(reply[0..sp.reply_size], std.mem.asBytes(&response));
|
||||||
|
return sp.reply_size;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
|
||||||
|
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
|
||||||
|
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
if (message.len < sp.request_size) return 0;
|
||||||
|
const request = std.mem.bytesToValue(sp.Request, message[0..sp.request_size]);
|
||||||
|
switch (request.operation) {
|
||||||
|
@intFromEnum(sp.Operation.present) => return scanoutStatus(reply, presentFull()),
|
||||||
|
@intFromEnum(sp.Operation.get_modes) => {
|
||||||
|
var response = sp.ModesReply{ .status = 0, .count = offered_modes.len, .modes = undefined };
|
||||||
|
for (0..sp.max_modes) |i| {
|
||||||
|
response.modes[i] = if (i < offered_modes.len)
|
||||||
|
.{ .width = offered_modes[i].width, .height = offered_modes[i].height }
|
||||||
|
else
|
||||||
|
.{ .width = 0, .height = 0 };
|
||||||
|
}
|
||||||
|
@memcpy(reply[0..sp.modes_reply_size], std.mem.asBytes(&response));
|
||||||
|
return sp.modes_reply_size;
|
||||||
|
},
|
||||||
|
@intFromEnum(sp.Operation.set_mode) => {
|
||||||
|
const w = request.width;
|
||||||
|
const h = request.height;
|
||||||
|
if (w == 0 or h == 0 or w > max_width or h > max_height) return scanoutStatus(reply, false);
|
||||||
|
current_width = w;
|
||||||
|
current_height = h;
|
||||||
|
return scanoutStatus(reply, setScanoutRect());
|
||||||
|
},
|
||||||
|
else => return 0,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse {
|
||||||
|
_ = system.write("virtio-gpu: missing device id (argv[1])\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
log("virtio-gpu: malformed device id '{s}'\n", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
runtime.service.run(256, .{
|
||||||
|
.service = .scanout,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
//! virtio 1.0 PCI transport — the vendor capabilities in PCI config space that point at the
|
||||||
|
//! device's structures (common config, notify, ISR) in a BAR, the common-config register
|
||||||
|
//! block, and the split-virtqueue layout. `extern` structs matching the spec. Host-tested
|
||||||
|
//! for size. See docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// PCI vendor-specific capability id (0x09) — virtio 1.0 structures are advertised as these.
|
||||||
|
pub const pci_cap_vendor: u8 = 0x09;
|
||||||
|
|
||||||
|
/// virtio_pci_cap `cfg_type`: which structure a vendor capability points at.
|
||||||
|
pub const cfg_common: u8 = 1;
|
||||||
|
pub const cfg_notify: u8 = 2;
|
||||||
|
pub const cfg_isr: u8 = 3;
|
||||||
|
pub const cfg_device: u8 = 4;
|
||||||
|
pub const cfg_pci: u8 = 5;
|
||||||
|
|
||||||
|
/// virtio_pci_cap — a vendor capability naming a structure at (bar, offset, length) within
|
||||||
|
/// a PCI BAR. Read straight out of config space.
|
||||||
|
pub const PciCap = extern struct {
|
||||||
|
cap_vndr: u8, // 0x09
|
||||||
|
cap_next: u8, // next capability's offset in config space (0 = end)
|
||||||
|
cap_len: u8,
|
||||||
|
cfg_type: u8, // cfg_common / cfg_notify / ...
|
||||||
|
bar: u8, // which BAR the structure lives in
|
||||||
|
padding: [3]u8,
|
||||||
|
offset: u32, // offset within the BAR
|
||||||
|
length: u32, // length of the structure
|
||||||
|
};
|
||||||
|
|
||||||
|
/// virtio_pci_notify_cap: a notify capability carries a multiplier after the base cap; the
|
||||||
|
/// per-queue notify address is `notify_base + queue_notify_off * notify_off_multiplier`.
|
||||||
|
pub const NotifyCap = extern struct {
|
||||||
|
cap: PciCap,
|
||||||
|
notify_off_multiplier: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// virtio_pci_common_cfg — the common configuration register block (little-endian MMIO).
|
||||||
|
pub const CommonCfg = extern struct {
|
||||||
|
device_feature_select: u32,
|
||||||
|
device_feature: u32,
|
||||||
|
driver_feature_select: u32,
|
||||||
|
driver_feature: u32,
|
||||||
|
msix_config: u16,
|
||||||
|
num_queues: u16,
|
||||||
|
device_status: u8,
|
||||||
|
config_generation: u8,
|
||||||
|
queue_select: u16,
|
||||||
|
queue_size: u16,
|
||||||
|
queue_msix_vector: u16,
|
||||||
|
queue_enable: u16,
|
||||||
|
queue_notify_off: u16,
|
||||||
|
queue_desc: u64,
|
||||||
|
queue_driver: u64,
|
||||||
|
queue_device: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// device_status bits (written to `CommonCfg.device_status` during bring-up).
|
||||||
|
pub const status_acknowledge: u8 = 1;
|
||||||
|
pub const status_driver: u8 = 2;
|
||||||
|
pub const status_driver_ok: u8 = 4;
|
||||||
|
pub const status_features_ok: u8 = 8;
|
||||||
|
|
||||||
|
/// VIRTIO_F_VERSION_1 — feature bit 32 (in the second 32-bit feature word). Required for a
|
||||||
|
/// modern device; we negotiate exactly this bit and nothing else.
|
||||||
|
pub const feature_version_1_word: u32 = 1; // device_feature_select value for bits 32..63
|
||||||
|
pub const feature_version_1_bit: u32 = 1 << 0; // bit 32 within that word
|
||||||
|
|
||||||
|
// --- split virtqueue -------------------------------------------------------
|
||||||
|
|
||||||
|
pub const Desc = extern struct {
|
||||||
|
addr: u64, // guest-physical
|
||||||
|
len: u32,
|
||||||
|
flags: u16,
|
||||||
|
next: u16,
|
||||||
|
};
|
||||||
|
pub const desc_flag_next: u16 = 1; // buffer continues in `next`
|
||||||
|
pub const desc_flag_write: u16 = 2; // device-writable (else driver-writable/device-readable)
|
||||||
|
|
||||||
|
/// The available ring's fixed header; a `[queue_size]u16` ring and a trailing `used_event`
|
||||||
|
/// u16 follow it in memory (laid out by the driver).
|
||||||
|
pub const AvailHdr = extern struct {
|
||||||
|
flags: u16,
|
||||||
|
idx: u16,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One entry of the used ring.
|
||||||
|
pub const UsedElem = extern struct {
|
||||||
|
id: u32,
|
||||||
|
len: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The used ring's fixed header; a `[queue_size]UsedElem` ring and a trailing `avail_event`
|
||||||
|
/// u16 follow it.
|
||||||
|
pub const UsedHdr = extern struct {
|
||||||
|
flags: u16,
|
||||||
|
idx: u16,
|
||||||
|
};
|
||||||
|
|
||||||
|
test "virtio-pci struct sizes match the spec" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(PciCap));
|
||||||
|
try std.testing.expectEqual(@as(usize, 56), @sizeOf(CommonCfg));
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Desc));
|
||||||
|
try std.testing.expectEqual(@as(usize, 8), @sizeOf(UsedElem));
|
||||||
|
}
|
||||||
@@ -106,6 +106,12 @@ pub fn serialWrite(bytes: []const u8) void {
|
|||||||
serial.write(bytes);
|
serial.write(bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether a working UART was detected (loopback probe). When false the serial
|
||||||
|
/// sink is silently inert — a dead legacy COM1 costs nothing per byte.
|
||||||
|
pub fn serialPresent() bool {
|
||||||
|
return serial.present();
|
||||||
|
}
|
||||||
|
|
||||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||||
/// displays. The last-resort progress signal when there's no text output at all.
|
/// displays. The last-resort progress signal when there's no text output at all.
|
||||||
@@ -162,10 +168,18 @@ pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, e
|
|||||||
paging.mapUserInto(root, virtual, physical, writable, executable);
|
paging.mapUserInto(root, virtual, physical, writable, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
/// Map a device MMIO window into address space `root`: RW+NX, and marked so teardown
|
||||||
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
/// won't free the MMIO frames as RAM. `write_combining` picks the cache type —
|
||||||
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
/// false = strong-uncacheable (registers), true = write-combining (a framebuffer).
|
||||||
paging.mapUserDeviceInto(root, virtual, physical, len);
|
/// For IO passthrough.
|
||||||
|
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64, write_combining: bool) void {
|
||||||
|
paging.mapUserDeviceInto(root, virtual, physical, len, write_combining);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is the user leaf mapping `virtual` in address space `root` write-combining? Null if
|
||||||
|
/// unmapped. For tests verifying the framebuffer map's cache type.
|
||||||
|
pub fn userLeafIsWriteCombining(root: u64, virtual: u64) ?bool {
|
||||||
|
return paging.leafIsWriteCombining(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map coherent DMA RAM into address space `root`: strong-uncacheable, RW+NX, but
|
/// Map coherent DMA RAM into address space `root`: strong-uncacheable, RW+NX, but
|
||||||
@@ -174,6 +188,13 @@ pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
|||||||
paging.mapUserDmaInto(root, virtual, physical, len);
|
paging.mapUserDmaInto(root, virtual, physical, len);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map shared cacheable RAM into address space `root`: write-back cacheable, RW+NX, and
|
||||||
|
/// marked so teardown won't free the frames (they're owned by a refcounted shm object,
|
||||||
|
/// freed when its last capability drops). For shm_create/shm_map.
|
||||||
|
pub fn mapUserSharedInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||||
|
paging.mapUserSharedInto(root, virtual, physical, len);
|
||||||
|
}
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||||
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||||
paging.map(virtual, physical, writable);
|
paging.map(virtual, physical, writable);
|
||||||
|
|||||||
@@ -7,8 +7,13 @@
|
|||||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||||
//! which the kernel heap will build on.
|
//! which the kernel heap will build on.
|
||||||
//!
|
//!
|
||||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
//! The physmap (the permanent window onto all physical RAM) is built with 2 MiB
|
||||||
//! negligible against available RAM.
|
//! huge pages wherever the range is 2 MiB-aligned, falling back to 4 KiB for the
|
||||||
|
//! unaligned edges. On a big machine that is the difference between ~16.7M page-
|
||||||
|
//! table entries (128 MiB of tables) and ~32K — it makes both the build and the
|
||||||
|
//! footprint scale sanely with RAM. Everything else (kernel segments, heap, user
|
||||||
|
//! space, on-demand MMIO) stays 4 KiB: precise, and the table memory is
|
||||||
|
//! negligible there.
|
||||||
|
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
@@ -22,10 +27,22 @@ const writable: u64 = 1 << 1;
|
|||||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||||
const pwt: u64 = 1 << 3; // page write-through
|
const pwt: u64 = 1 << 3; // page write-through
|
||||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||||
|
const page_size_bit: u64 = 1 << 7; // PS: this PDPT/PD entry is a 1 GiB/2 MiB leaf, not a pointer to the next table
|
||||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||||
const no_execute: u64 = 1 << 63;
|
const no_execute: u64 = 1 << 63;
|
||||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
|
// The PAT-index bit. In a 4 KiB PTE it is bit 7; in a huge leaf (2 MiB PDE / 1 GiB
|
||||||
|
// PDPTE) bit 7 is PS, so the PAT bit moves to bit 12. With PCD=PWT=0 this selects
|
||||||
|
// PAT entry 4, which `setupPat` programs to write-combining (see mapRangePhysmap).
|
||||||
|
const pte_pat: u64 = 1 << 7;
|
||||||
|
const huge_pat: u64 = 1 << 12;
|
||||||
|
const ia32_pat: u32 = 0x277;
|
||||||
|
|
||||||
|
/// The physmap's page size for 2 MiB-aligned RAM: one PD leaf covers this instead
|
||||||
|
/// of 512 PT entries. 4 KiB pages fill the unaligned edges (see mapRangePhysmap).
|
||||||
|
const huge_page_size: u64 = 2 << 20; // 2 MiB
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
// ELF segment flags (p_flags).
|
||||||
const pf_x: u32 = 1;
|
const pf_x: u32 = 1;
|
||||||
const pf_w: u32 = 2;
|
const pf_w: u32 = 2;
|
||||||
@@ -74,7 +91,15 @@ fn allocTable() u64 {
|
|||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||||
/// writable only if every level is; non-executable if any level is).
|
/// writable only if every level is; non-executable if any level is).
|
||||||
fn descend(entry: *u64) u64 {
|
fn descend(entry: *u64) u64 {
|
||||||
if (entry.* & present != 0) return entry.* & address_mask;
|
if (entry.* & present != 0) {
|
||||||
|
// A present-but-huge entry is a leaf, not a table: descending would read
|
||||||
|
// its 2 MiB/1 GiB data frame as a page table and corrupt RAM. This only
|
||||||
|
// fires on a bug — a 4 KiB map landing inside a physmap huge page — and a
|
||||||
|
// loud panic beats silent corruption. (The physmap and the 4 KiB regions
|
||||||
|
// live in disjoint PML4 slots, so it should never happen.)
|
||||||
|
if (entry.* & page_size_bit != 0) @panic("paging: descend through a huge-page leaf");
|
||||||
|
return entry.* & address_mask;
|
||||||
|
}
|
||||||
const frame = allocTable();
|
const frame = allocTable();
|
||||||
entry.* = frame | present | writable;
|
entry.* = frame | present | writable;
|
||||||
return frame;
|
return frame;
|
||||||
@@ -96,15 +121,39 @@ fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
|||||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map one 2 MiB huge page `virtual` -> `physical` with `flags` — a leaf at the PD
|
||||||
|
/// level (PS bit set), with no PT beneath it. Both addresses must be 2 MiB-aligned.
|
||||||
|
/// One of these replaces 512 `mapPage`s (and the PT frame they'd need).
|
||||||
|
fn mapHugePage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||||
|
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||||
|
@panic("paging: new higher-half PML4 entry after init");
|
||||||
|
const pdpt = descend(pml4e);
|
||||||
|
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = descend(pdpte);
|
||||||
|
tableAt(pd)[(virtual >> 21) & 0x1FF] = (physical & address_mask) | flags | present | page_size_bit;
|
||||||
|
}
|
||||||
|
|
||||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||||
/// window onto physical memory once the low identity map goes away.
|
/// window onto physical memory once the low identity map goes away. The 2 MiB-
|
||||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
/// aligned interior is mapped with huge pages; the unaligned head/tail with 4 KiB.
|
||||||
|
/// `write_combining` selects the WC memory type (setupPat's PAT entry 4) via the
|
||||||
|
/// PAT bit — bit 7 in a 4 KiB PTE, bit 12 in a huge leaf — for the framebuffer.
|
||||||
|
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64, write_combining: bool) void {
|
||||||
|
const pte_flags = if (write_combining) flags | pte_pat else flags;
|
||||||
|
const huge_flags = if (write_combining) flags | huge_pat else flags;
|
||||||
var address = physical_base & ~@as(u64, page_size - 1);
|
var address = physical_base & ~@as(u64, page_size - 1);
|
||||||
const end = physical_base + len;
|
const end = physical_base + len;
|
||||||
while (address < end) : (address += page_size) {
|
// Head: 4 KiB pages up to the next 2 MiB boundary.
|
||||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
while (address < end and address & (huge_page_size - 1) != 0) : (address += page_size)
|
||||||
}
|
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||||
|
// Interior: 2 MiB huge pages while a whole one still fits.
|
||||||
|
while (address + huge_page_size <= end) : (address += huge_page_size)
|
||||||
|
mapHugePage(pml4, boot_handoff.physicalToVirtual(address), address, huge_flags);
|
||||||
|
// Tail: 4 KiB pages for whatever is left.
|
||||||
|
while (address < end) : (address += page_size)
|
||||||
|
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||||
@@ -118,11 +167,26 @@ fn enableNx() void {
|
|||||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Program this core's PAT so entry 4 (selected by the PAT bit with PCD=PWT=0) is
|
||||||
|
/// **write-combining**, leaving the other seven at their reset types. Nothing else
|
||||||
|
/// in danos sets the PAT bit, so this changes no existing mapping — it only gives
|
||||||
|
/// the framebuffer a write-combining type, which turns its full-screen clear from
|
||||||
|
/// glacial (uncached writes to a GPU BAR, the real-hardware default via MTRRs) into
|
||||||
|
/// a batched burst. Must run on **every** core (PAT is per-logical-processor) — the
|
||||||
|
/// framebuffer mapping lives in the shared kernel half, so a core with the reset
|
||||||
|
/// PAT would see it as write-back and alias. Called from `init` (BSP) and each AP.
|
||||||
|
pub fn setupPat() void {
|
||||||
|
// Reset PAT is PA0=WB PA1=WT PA2=UC- PA3=UC PA4=WB PA5=WT PA6=UC- PA7=UC; flip
|
||||||
|
// PA4 from WB (0x06) to WC (0x01). Type codes: UC=0 WC=1 WT=4 WP=5 WB=6 UC-=7.
|
||||||
|
io.wrmsr(ia32_pat, 0x0007_0401_0007_0406);
|
||||||
|
}
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
/// Build the address space and switch onto it.
|
||||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||||
alloc_frame = allocFrame;
|
alloc_frame = allocFrame;
|
||||||
free_frame = freeFrame;
|
free_frame = freeFrame;
|
||||||
enableNx();
|
enableNx();
|
||||||
|
setupPat(); // BSP: PAT entry 4 = write-combining, for the framebuffer window
|
||||||
const pml4 = allocTable();
|
const pml4 = allocTable();
|
||||||
|
|
||||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||||
@@ -130,13 +194,15 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
|||||||
// mapped on demand (mapMmio) or explicitly below.
|
// mapped on demand (mapMmio) or explicitly below.
|
||||||
for (regions(boot_information.memory_map)) |r| {
|
for (regions(boot_information.memory_map)) |r| {
|
||||||
if (r.kind == .mmio) continue;
|
if (r.kind == .mmio) continue;
|
||||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||||
// the kernel touches directly), RW + NX.
|
// the kernel touches directly), RW + NX. The framebuffer is **write-
|
||||||
|
// combining** (see setupPat) so the console's full-screen clear is a burst,
|
||||||
|
// not millions of uncached single-word writes.
|
||||||
const fb = boot_information.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute, true);
|
||||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||||
|
|
||||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||||
@@ -254,8 +320,12 @@ pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool,
|
|||||||
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
||||||
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
||||||
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
||||||
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64, write_combining: bool) void {
|
||||||
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
// Registers are strong-uncacheable (PCD|PWT). A framebuffer instead wants
|
||||||
|
// write-combining — the PAT bit (bit 7 in a 4 KiB PTE) with PCD=PWT=0 selects PAT
|
||||||
|
// entry 4, which `setupPat` programs to WC — so pixel writes batch into bursts.
|
||||||
|
const cache: u64 = if (write_combining) pte_pat else (pcd | pwt);
|
||||||
|
const flags: u64 = present | user | writable | no_execute | device_grant | cache;
|
||||||
const first = physical & ~@as(u64, page_size - 1);
|
const first = physical & ~@as(u64, page_size - 1);
|
||||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||||
var off: u64 = 0;
|
var off: u64 = 0;
|
||||||
@@ -296,6 +366,57 @@ pub fn mapUserDmaInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The raw leaf entry mapping `virtual` in the address space rooted at `pml4`, or null
|
||||||
|
/// if any level of the walk is absent. **Read-only** — never allocates or descends into
|
||||||
|
/// a missing table (unlike the `map*` paths' `descendUser`). Stops at the first huge
|
||||||
|
/// leaf. For tests and introspection that need a page's actual flag bits.
|
||||||
|
pub fn leafEntryOf(pml4: u64, virtual: u64) ?u64 {
|
||||||
|
const l4 = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (l4 & present == 0) return null;
|
||||||
|
const l3 = tableAt(l4 & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
|
if (l3 & present == 0) return null;
|
||||||
|
if (l3 & page_size_bit != 0) return l3; // 1 GiB leaf
|
||||||
|
const l2 = tableAt(l3 & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
|
if (l2 & present == 0) return null;
|
||||||
|
if (l2 & page_size_bit != 0) return l2; // 2 MiB leaf
|
||||||
|
const l1 = tableAt(l2 & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
|
if (l1 & present == 0) return null;
|
||||||
|
return l1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is the 4 KiB leaf mapping `virtual` write-combining — the PAT bit set with PCD and
|
||||||
|
/// PWT clear, which `setupPat` makes PAT entry 4 (WC)? Null if unmapped. The device
|
||||||
|
/// mapping path (`mapUserDeviceInto`) always uses 4 KiB leaves, so bit 7 (`pte_pat`)
|
||||||
|
/// is the PAT selector in play.
|
||||||
|
pub fn leafIsWriteCombining(pml4: u64, virtual: u64) ?bool {
|
||||||
|
const e = leafEntryOf(pml4, virtual) orelse return null;
|
||||||
|
return (e & pte_pat != 0) and (e & pcd == 0) and (e & pwt == 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map `[physical, physical+len)` into the user half rooted at `pml4` as **shared cacheable
|
||||||
|
/// RAM**: write-back cacheable (RW + NX) for CPU compositing, and carrying `device_grant`
|
||||||
|
/// so teardown (`freeSubtree`) does **not** return the frames to the allocator. The frames
|
||||||
|
/// are owned by a refcounted shared-memory object (system/kernel/ipc-synchronous.zig) and
|
||||||
|
/// freed only when its last capability drops — not when one sharer's address space dies, or
|
||||||
|
/// the other sharers would be left mapping freed RAM. The caller aligns `virtual`/`physical`.
|
||||||
|
pub fn mapUserSharedInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||||
|
const flags: u64 = present | user | writable | no_execute | device_grant; // WB cacheable
|
||||||
|
const first = physical & ~@as(u64, page_size - 1);
|
||||||
|
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||||
|
var off: u64 = 0;
|
||||||
|
while (first + off <= last) : (off += page_size) {
|
||||||
|
const v = virtual + off;
|
||||||
|
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||||
|
const pdpt = descendUser(pml4e);
|
||||||
|
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||||
|
const pd = descendUser(pdpte);
|
||||||
|
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||||
|
const pt = descendUser(pde);
|
||||||
|
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||||
|
invalidate(v);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Create a new address space: a fresh PML4 with an empty user half and the
|
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||||
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||||
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||||
@@ -338,8 +459,8 @@ fn freeSubtree(physical: u64, level: u32) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
/// clear. Walks the 4-level tables, stopping at a 2 MiB huge-page leaf (the physmap
|
||||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
/// uses them). Returns false if unmapped. Used for W^X checks in tests.
|
||||||
pub fn isExecutable(virtual: u64) bool {
|
pub fn isExecutable(virtual: u64) bool {
|
||||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return false;
|
if (pml4e & present == 0) return false;
|
||||||
@@ -347,6 +468,7 @@ pub fn isExecutable(virtual: u64) bool {
|
|||||||
if (pdpte & present == 0) return false;
|
if (pdpte & present == 0) return false;
|
||||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return false;
|
if (pde & present == 0) return false;
|
||||||
|
if (pde & page_size_bit != 0) return pde & no_execute == 0; // 2 MiB huge leaf
|
||||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return false;
|
if (pte & present == 0) return false;
|
||||||
return pte & no_execute == 0;
|
return pte & no_execute == 0;
|
||||||
@@ -385,9 +507,9 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
|||||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||||
/// needs the frame behind a user vaddr to free it).
|
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
||||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return null;
|
if (pml4e & present == 0) return null;
|
||||||
@@ -395,6 +517,8 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
|||||||
if (pdpte & present == 0) return null;
|
if (pdpte & present == 0) return null;
|
||||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return null;
|
if (pde & present == 0) return null;
|
||||||
|
if (pde & page_size_bit != 0) // 2 MiB huge leaf: frame base is bits 51:21
|
||||||
|
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return null;
|
if (pte & present == 0) return null;
|
||||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
|
|||||||
@@ -17,6 +17,12 @@ const Access = enum { port, mmio };
|
|||||||
var access: Access = .port;
|
var access: Access = .port;
|
||||||
var base: u64 = 0x3F8; // COM1
|
var base: u64 = 0x3F8; // COM1
|
||||||
|
|
||||||
|
/// Whether `init`/`reconfigure` found a *working* UART at `base`. False on a
|
||||||
|
/// legacy-free machine whose COM1 is decoded but dead: writing to it is then a
|
||||||
|
/// no-op, so `write` never spins waiting for a transmit register that will never
|
||||||
|
/// drain. Cleared until proven by the loopback probe.
|
||||||
|
var uart_present: bool = false;
|
||||||
|
|
||||||
fn portOut(p: u16, value: u8) void {
|
fn portOut(p: u16, value: u8) void {
|
||||||
asm volatile ("outb %[value], %[p]"
|
asm volatile ("outb %[value], %[p]"
|
||||||
:
|
:
|
||||||
@@ -57,6 +63,34 @@ pub fn init() void {
|
|||||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||||
setRegister(4, 0x0B); // RTS/DSR set
|
setRegister(4, 0x0B); // RTS/DSR set
|
||||||
|
uart_present = probe();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect a *working* UART by internal loopback: route the transmitter back to
|
||||||
|
/// the receiver (MCR bit 4), send a byte, and check it comes back. A port that is
|
||||||
|
/// merely decoded but has nothing behind it (the common case on a legacy-free
|
||||||
|
/// board that still answers I/O at 0x3F8) never echoes, so this returns false.
|
||||||
|
///
|
||||||
|
/// This matters for speed, not just correctness: a dead UART's line-status
|
||||||
|
/// register reads back 0x00, so its transmit-holding-empty bit never sets, and
|
||||||
|
/// `writeByte` would otherwise spin its full guard — tens of milliseconds — on
|
||||||
|
/// *every* logged byte. On real hardware that alone can add ~a minute to boot.
|
||||||
|
fn probe() bool {
|
||||||
|
const saved_mcr = register(4);
|
||||||
|
setRegister(4, 0x1E); // MCR: LOOP | OUT2 | OUT1 | RTS — internal loopback
|
||||||
|
setRegister(0, 0xAE); // push a distinctive byte into the loopback path
|
||||||
|
var guard: u32 = 0;
|
||||||
|
while (register(5) & 0x01 == 0 and guard < 10_000) : (guard += 1) {} // await Data Ready
|
||||||
|
const echo = register(0);
|
||||||
|
setRegister(4, saved_mcr); // restore the modem-control lines
|
||||||
|
return echo == 0xAE;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a working UART was detected (see `probe`). The log sink stays
|
||||||
|
/// registered regardless — it simply does nothing until this is true — so a UART
|
||||||
|
/// that only `reconfigure` discovers (via SPCR) still starts logging.
|
||||||
|
pub fn present() bool {
|
||||||
|
return uart_present;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||||
@@ -70,15 +104,19 @@ pub fn reconfigure(is_mmio: bool, address: u64) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn writeByte(c: u8) void {
|
fn writeByte(c: u8) void {
|
||||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
// Wait for the transmit-holding register to empty. `write` only reaches here
|
||||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
// for a UART the loopback probe proved live, so this bounds a momentary stall
|
||||||
|
// (e.g. deasserted flow control), not an absent port: ~5000 legacy-port reads
|
||||||
|
// is a few ms — comfortably longer than one 38400-baud byte-time (~260 µs).
|
||||||
var guard: u32 = 0;
|
var guard: u32 = 0;
|
||||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
while (register(5) & 0x20 == 0 and guard < 5_000) : (guard += 1) {}
|
||||||
setRegister(0, c);
|
setRegister(0, c);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
/// Write bytes, translating LF to CRLF so terminals and logs line up. A no-op
|
||||||
|
/// when no working UART was detected, so a dead COM1 costs nothing per byte.
|
||||||
pub fn write(bytes: []const u8) void {
|
pub fn write(bytes: []const u8) void {
|
||||||
|
if (!uart_present) return;
|
||||||
for (bytes) |c| {
|
for (bytes) |c| {
|
||||||
if (c == '\n') writeByte('\r');
|
if (c == '\n') writeByte('\r');
|
||||||
writeByte(c);
|
writeByte(c);
|
||||||
|
|||||||
@@ -173,6 +173,7 @@ fn delayMicros(us: u64) void {
|
|||||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||||
const cpu = boot_index;
|
const cpu = boot_index;
|
||||||
|
paging.setupPat(); // this core's PAT: entry 4 = write-combining, to match the BSP
|
||||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||||
idt.loadOnThisCpu(); // the shared IDT
|
idt.loadOnThisCpu(); // the shared IDT
|
||||||
|
|||||||
@@ -2,12 +2,15 @@
|
|||||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||||
//! — just pixels.
|
//! — just pixels.
|
||||||
//!
|
//!
|
||||||
//! This is a **bootstrap** console — a stop-gap so early boot has something on
|
//! This is a **bootstrap / fatal-fallback** console. The driver machinery now exists — the
|
||||||
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
|
//! user-space **display service** ([../services/display](../services/display/display.zig),
|
||||||
//! terminal; once the driver machinery exists it becomes a proper graphics device
|
//! docs/display.md) owns the framebuffer in normal operation — so this no longer paints
|
||||||
//! driver and this text-grid crutch goes away. It is therefore kept **separate
|
//! routine status. It exists for the two cases the display service can't cover: **early
|
||||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
|
//! boot**, before the service has claimed the framebuffer, and **fatal errors** (a kernel
|
||||||
//! while this only paints the handful of user-facing status lines and panics.
|
//! panic or a kernel-mode fault), which force it back on (`setSuppressed`) so a dying
|
||||||
|
//! machine's last words reach the screen even over a live display. It is kept **separate
|
||||||
|
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file and
|
||||||
|
//! carries all routine kernel output; this only paints those fatal cases.
|
||||||
//!
|
//!
|
||||||
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
||||||
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
||||||
@@ -20,6 +23,12 @@ const boot_handoff = @import("boot-handoff");
|
|||||||
var con: Console = undefined;
|
var con: Console = undefined;
|
||||||
var con_present: bool = false;
|
var con_present: bool = false;
|
||||||
|
|
||||||
|
/// Set while a user-space display service owns the framebuffer: `write` falls silent so
|
||||||
|
/// the kernel doesn't paint over the compositor. Driven by the display device's
|
||||||
|
/// claim/release (system/kernel/process.zig). The terminal panic/exception paths clear
|
||||||
|
/// it first (`setSuppressed(false)`) — a dying machine's message wins over any display.
|
||||||
|
var suppressed: bool = false;
|
||||||
|
|
||||||
/// Set up the console over `fb`, or mark it absent if there's no usable
|
/// Set up the console over `fb`, or mark it absent if there's no usable
|
||||||
/// framebuffer. Clears the screen when present.
|
/// framebuffer. Clears the screen when present.
|
||||||
pub fn init(fb: boot_handoff.Framebuffer) void {
|
pub fn init(fb: boot_handoff.Framebuffer) void {
|
||||||
@@ -41,13 +50,20 @@ pub fn present() bool {
|
|||||||
return con_present;
|
return con_present;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present,
|
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present, or
|
||||||
/// so it's always safe to call.
|
/// while a display service owns the screen (`suppressed`), so it's always safe to call.
|
||||||
pub fn write(bytes: []const u8) void {
|
pub fn write(bytes: []const u8) void {
|
||||||
if (!con_present) return;
|
if (!con_present or suppressed) return;
|
||||||
for (bytes) |c| con.putChar(c);
|
for (bytes) |c| con.putChar(c);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Quiesce (or resume) the bootstrap console. Set true when a display service claims the
|
||||||
|
/// framebuffer; set false when that claim is released, or by the panic path to force a
|
||||||
|
/// last message onto a screen a (now-irrelevant) service was holding.
|
||||||
|
pub fn setSuppressed(value: bool) void {
|
||||||
|
suppressed = value;
|
||||||
|
}
|
||||||
|
|
||||||
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
||||||
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
||||||
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
||||||
|
|||||||
@@ -36,6 +36,11 @@ var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined;
|
|||||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||||
var count: usize = 0;
|
var count: usize = 0;
|
||||||
|
|
||||||
|
/// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine
|
||||||
|
/// handed over no framebuffer. Lets the process layer recognise the display claim
|
||||||
|
/// (to quiesce the bootstrap console) without threading the id through every caller.
|
||||||
|
var display_device: ?u64 = null;
|
||||||
|
|
||||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||||
@@ -45,10 +50,55 @@ pub var dropped: usize = 0;
|
|||||||
pub fn init(device_tree: *const platform.DeviceTree) void {
|
pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||||
count = 0;
|
count = 0;
|
||||||
dropped = 0;
|
dropped = 0;
|
||||||
|
display_device = null;
|
||||||
for (&claimed) |*c| c.* = null;
|
for (&claimed) |*c| c.* = null;
|
||||||
walk(device_tree.root, device_abi.no_parent);
|
walk(device_tree.root, device_abi.no_parent);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Publish the loader's framebuffer as a `display` device — a root-level node with one
|
||||||
|
/// write-combining `memory` resource over the linear framebuffer and its geometry in
|
||||||
|
/// `.display`. The framebuffer is *not* firmware-discovered (it rides the
|
||||||
|
/// [[boot-handoff]], not the device tree), so it is seeded explicitly, after `init`.
|
||||||
|
/// Returns the new device id, or null when there is no framebuffer (headless) or the
|
||||||
|
/// table is full. Idempotent-ish: only ever call once per boot.
|
||||||
|
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32) ?u64 {
|
||||||
|
if (base == 0 or width == 0 or height == 0) return null; // headless
|
||||||
|
if (count >= maximum_devices) {
|
||||||
|
dropped += 1;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||||
|
d.id = count;
|
||||||
|
d.parent = device_abi.no_parent;
|
||||||
|
d.class = @intFromEnum(device_abi.DeviceClass.display);
|
||||||
|
d.pci_class = device_abi.no_pci_class;
|
||||||
|
d.resource_count = 1;
|
||||||
|
d.resources[0] = .{
|
||||||
|
.kind = @intFromEnum(device_abi.ResourceKind.memory),
|
||||||
|
.start = base,
|
||||||
|
.len = @as(u64, height) * pitch,
|
||||||
|
.flags = device_abi.resource_flag_write_combining,
|
||||||
|
};
|
||||||
|
d.display = .{ .width = width, .height = height, .pitch = pitch, .format = format };
|
||||||
|
devices[count] = d;
|
||||||
|
display_device = d.id;
|
||||||
|
count += 1;
|
||||||
|
return d.id;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The id of the seeded framebuffer device, or null when none was seeded.
|
||||||
|
pub fn displayDevice() ?u64 {
|
||||||
|
return display_device;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the framebuffer device is currently claimed by some process. The bootstrap
|
||||||
|
/// console uses this (via the process layer) to fall silent while a display service
|
||||||
|
/// owns the screen, and to resume if that service dies and its claim is released.
|
||||||
|
pub fn displayClaimed() bool {
|
||||||
|
const id = display_device orelse return false;
|
||||||
|
return ownerOf(id) != null;
|
||||||
|
}
|
||||||
|
|
||||||
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||||
/// assigned it down to its children as their parent.
|
/// assigned it down to its children as their parent.
|
||||||
fn walk(node: *platform.Device, parent_id: u64) void {
|
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ const architecture = @import("architecture");
|
|||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
|
const pmm = @import("pmm.zig");
|
||||||
|
|
||||||
const page_size = abi.page_size;
|
const page_size = abi.page_size;
|
||||||
const Task = scheduler.Task;
|
const Task = scheduler.Task;
|
||||||
@@ -89,6 +90,10 @@ const user_half_end: u64 = 0x0000_8000_0000_0000;
|
|||||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||||
pub const Endpoint = struct {
|
pub const Endpoint = struct {
|
||||||
refcount: u32 = 1,
|
refcount: u32 = 1,
|
||||||
|
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||||
|
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||||
|
owner: u32 = 0,
|
||||||
|
dead: bool = false,
|
||||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||||
sender_head: ?*Task = null,
|
sender_head: ?*Task = null,
|
||||||
@@ -109,10 +114,30 @@ pub const Endpoint = struct {
|
|||||||
|
|
||||||
pub fn createIpcEndpoint() ?*Endpoint {
|
pub fn createIpcEndpoint() ?*Endpoint {
|
||||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||||
endpoint.* = .{};
|
endpoint.* = .{ .owner = scheduler.currentId() };
|
||||||
return endpoint;
|
return endpoint;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
||||||
|
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
||||||
|
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
||||||
|
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
||||||
|
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
||||||
|
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||||
|
for (®istry) |*slot| {
|
||||||
|
const endpoint = slot.* orelse continue;
|
||||||
|
if (endpoint.owner != task_id) continue;
|
||||||
|
endpoint.dead = true;
|
||||||
|
while (dequeueSender(endpoint)) |sender| {
|
||||||
|
sender.ipc_status = -EPEER;
|
||||||
|
sender.ipc_received_cap = abi.no_cap;
|
||||||
|
scheduler.readyLocked(sender);
|
||||||
|
}
|
||||||
|
slot.* = null;
|
||||||
|
dropRef(endpoint);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||||
pub fn dropRef(endpoint: *Endpoint) void {
|
pub fn dropRef(endpoint: *Endpoint) void {
|
||||||
@@ -123,6 +148,43 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- capability objects: what a handle-table entry can name ------------------
|
||||||
|
|
||||||
|
/// The `kind` tag on a `scheduler.HandleObject` — which capability object a handle names.
|
||||||
|
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||||
|
pub const handle_kind_endpoint: u8 = 0;
|
||||||
|
pub const handle_kind_shm: u8 = 1;
|
||||||
|
|
||||||
|
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||||
|
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||||
|
/// contiguous physical base, `pages` its length. A sharer's address-space teardown never
|
||||||
|
/// reclaims these frames (the mapping carries `device_grant`); this object owns them.
|
||||||
|
pub const ShmObject = struct {
|
||||||
|
refcount: u32 = 1,
|
||||||
|
phys: u64,
|
||||||
|
pages: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Wrap `pages` contiguous frames at `phys` (already allocated + zeroed by the caller) in a
|
||||||
|
/// refcounted shm object, or null if the heap is out of room.
|
||||||
|
pub fn createShm(phys: u64, pages: usize) ?*ShmObject {
|
||||||
|
const shm = heap.allocator().create(ShmObject) catch return null;
|
||||||
|
shm.* = .{ .phys = phys, .pages = pages };
|
||||||
|
return shm;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop a shared-memory reference; when the last one goes, return its frames to the
|
||||||
|
/// allocator and free the object. (The mappings themselves are torn down with each
|
||||||
|
/// sharer's address space; `device_grant` keeps that from freeing the frames early.)
|
||||||
|
pub fn dropShmRef(shm: *ShmObject) void {
|
||||||
|
if (shm.refcount > 1) {
|
||||||
|
shm.refcount -= 1;
|
||||||
|
} else {
|
||||||
|
for (0..shm.pages) |i| pmm.free(shm.phys + i * page_size);
|
||||||
|
heap.allocator().destroy(shm);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||||
|
|
||||||
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||||
@@ -222,11 +284,25 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
|||||||
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
|
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
|
||||||
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
|
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
|
||||||
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||||
const endpoint = resolveHandle(from, cap) orelse return -EBADF;
|
if (cap >= from.handles.len) return -EBADF;
|
||||||
endpoint.refcount += 1;
|
const entry = from.handles[@intCast(cap)] orelse return -EBADF;
|
||||||
const handle = installHandle(to, endpoint);
|
// Bump the named object's refcount (a copy, not a move — the sender keeps its handle),
|
||||||
|
// dispatching by kind so both endpoints and shared-memory regions can travel with a
|
||||||
|
// message.
|
||||||
|
switch (entry.kind) {
|
||||||
|
handle_kind_endpoint => {
|
||||||
|
const e: *Endpoint = @ptrCast(@alignCast(entry.ptr));
|
||||||
|
e.refcount += 1;
|
||||||
|
},
|
||||||
|
handle_kind_shm => {
|
||||||
|
const s: *ShmObject = @ptrCast(@alignCast(entry.ptr));
|
||||||
|
s.refcount += 1;
|
||||||
|
},
|
||||||
|
else => return -EBADF,
|
||||||
|
}
|
||||||
|
const handle = installEntry(to, entry);
|
||||||
if (handle < 0) {
|
if (handle < 0) {
|
||||||
dropRef(endpoint); // undo the bump; the receiver had no room
|
dropEntry(entry); // undo the bump; the receiver had no room
|
||||||
return -ENOSPC;
|
return -ENOSPC;
|
||||||
}
|
}
|
||||||
return handle;
|
return handle;
|
||||||
@@ -241,6 +317,7 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
|
|||||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
if (endpoint.dead) return -EPEER; // the service that owned this endpoint is gone — don't block
|
||||||
|
|
||||||
const me = scheduler.current();
|
const me = scheduler.current();
|
||||||
me.ipc_send_ptr = message_ptr;
|
me.ipc_send_ptr = message_ptr;
|
||||||
@@ -416,36 +493,68 @@ pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
|||||||
|
|
||||||
// --- per-process handle table + name registry -------------------------------
|
// --- per-process handle table + name registry -------------------------------
|
||||||
|
|
||||||
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
/// Install a capability object (kind + pointer) in task `t`'s handle table; returns the
|
||||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
/// small-int handle or -ENOSPC. The caller has already taken/holds the reference the slot
|
||||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
/// represents.
|
||||||
|
fn installEntry(t: *Task, entry: scheduler.HandleObject) i64 {
|
||||||
for (&t.handles, 0..) |*slot, i| {
|
for (&t.handles, 0..) |*slot, i| {
|
||||||
if (slot.* == null) {
|
if (slot.* == null) {
|
||||||
slot.* = @ptrCast(endpoint);
|
slot.* = entry;
|
||||||
return @intCast(i);
|
return @intCast(i);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return -ENOSPC;
|
return -ENOSPC;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
/// Install an endpoint handle. The common case; keeps the endpoint callers' signature.
|
||||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||||
if (h >= t.handles.len) return null;
|
return installEntry(t, .{ .kind = handle_kind_endpoint, .ptr = @ptrCast(endpoint) });
|
||||||
const slot = t.handles[@intCast(h)] orelse return null;
|
|
||||||
return @ptrCast(@alignCast(slot));
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
/// Install a shared-memory handle.
|
||||||
/// exit path so a dead server's endpoints don't linger referenced.
|
pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 {
|
||||||
|
return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
|
||||||
|
/// (e.g. an shm handle used where an endpoint is expected).
|
||||||
|
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||||
|
if (h >= t.handles.len) return null;
|
||||||
|
const entry = t.handles[@intCast(h)] orelse return null;
|
||||||
|
if (entry.kind != handle_kind_endpoint) return null;
|
||||||
|
return @ptrCast(@alignCast(entry.ptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve a handle to its shared-memory object, or null if out of range, unused, or not
|
||||||
|
/// an shm handle.
|
||||||
|
pub fn resolveShm(t: *Task, h: u64) ?*ShmObject {
|
||||||
|
if (h >= t.handles.len) return null;
|
||||||
|
const entry = t.handles[@intCast(h)] orelse return null;
|
||||||
|
if (entry.kind != handle_kind_shm) return null;
|
||||||
|
return @ptrCast(@alignCast(entry.ptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop every capability reference an exiting task holds, dispatching by kind so a dead
|
||||||
|
/// task's endpoints *and* shared-memory regions are released correctly. Called from the
|
||||||
|
/// scheduler exit path.
|
||||||
pub fn closeHandles(t: *Task) void {
|
pub fn closeHandles(t: *Task) void {
|
||||||
for (&t.handles) |*slot| {
|
for (&t.handles) |*slot| {
|
||||||
if (slot.*) |p| {
|
if (slot.*) |entry| {
|
||||||
dropRef(@ptrCast(@alignCast(p)));
|
dropEntry(entry);
|
||||||
slot.* = null;
|
slot.* = null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Drop the reference a handle-table entry represents, by kind.
|
||||||
|
fn dropEntry(entry: scheduler.HandleObject) void {
|
||||||
|
switch (entry.kind) {
|
||||||
|
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||||
|
handle_kind_shm => dropShmRef(@ptrCast(@alignCast(entry.ptr))),
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||||
|
|
||||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||||
|
|||||||
+80
-29
@@ -9,6 +9,7 @@ const wall_clock = @import("wall-clock.zig");
|
|||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
|
const sync = @import("sync.zig");
|
||||||
const process = @import("process.zig");
|
const process = @import("process.zig");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
const irq = @import("irq.zig");
|
const irq = @import("irq.zig");
|
||||||
@@ -60,8 +61,16 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||||
// with port-0x80 checkpoints as the only progress signal.
|
// with port-0x80 checkpoints as the only progress signal.
|
||||||
architecture.serialInit();
|
//
|
||||||
log.addSink(architecture.serialWrite);
|
// Serial is compiled in only under -Dserial (build.zig): a real machine often
|
||||||
|
// has no live legacy COM1, and the log survives in the RAM buffer (below) and
|
||||||
|
// is flushed to disk — so serial is now a QEMU/dev convenience the flashable
|
||||||
|
// image leaves out. When it *is* built in, `serialInit`'s loopback probe still
|
||||||
|
// guards against a dead port (so a -Dserial image is safe on real hardware).
|
||||||
|
if (build_options.serial) {
|
||||||
|
architecture.serialInit();
|
||||||
|
log.addSink(architecture.serialWrite);
|
||||||
|
}
|
||||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||||
// Retain the whole stream in a RAM buffer too, so a user program can later
|
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||||
// read it back (klog_read) and persist the boot log to disk — the only way to
|
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||||
@@ -72,8 +81,14 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||||
|
//
|
||||||
|
// The console is brought up *after* paging (below), not here: its one-time
|
||||||
|
// full-screen clear then runs on the kernel's **write-combining** mapping of the
|
||||||
|
// framebuffer instead of the loader's uncached one — a fast burst rather than
|
||||||
|
// millions of uncached writes on real hardware. Until then, on-screen output is
|
||||||
|
// absent (an early panic still lands in the serial/RAM log); the trade is worth
|
||||||
|
// a near-instant boot. `console.write` is a safe no-op while the console is down.
|
||||||
const fb = boot_information.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
console.init(fb);
|
|
||||||
|
|
||||||
log.checkpoint(cp_entry);
|
log.checkpoint(cp_entry);
|
||||||
|
|
||||||
@@ -83,10 +98,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
architecture.init();
|
architecture.init();
|
||||||
|
|
||||||
status("/system/kernel: initialising kernel...\n");
|
status("/system/kernel: initialising kernel...\n");
|
||||||
log.write(if (console.present())
|
if (build_options.serial) log.write(if (architecture.serialPresent())
|
||||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
"/system/kernel: serial console online (COM1)\n"
|
||||||
else
|
else
|
||||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
"/system/kernel: no serial UART (COM1 absent) -> log kept in RAM/debugcon\n");
|
||||||
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||||
@@ -142,6 +157,16 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||||
|
|
||||||
|
// Now on our own tables, the framebuffer window is write-combining: bring up the
|
||||||
|
// on-screen console and clear it to a blank canvas (a fast burst here, not the loader's
|
||||||
|
// uncached crawl). Routine boot output goes only to the log; this console now exists for
|
||||||
|
// early-boot and fatal (`fatal`/panic) output, until the display service takes over.
|
||||||
|
console.init(fb);
|
||||||
|
log.write(if (console.present())
|
||||||
|
"/system/kernel: framebuffer ready (early-boot + fatal fallback; the display service drives it in normal operation)\n"
|
||||||
|
else
|
||||||
|
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||||
heap.init();
|
heap.init();
|
||||||
log.checkpoint(cp_heap);
|
log.checkpoint(cp_heap);
|
||||||
@@ -172,25 +197,24 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
log.print("/system/kernel: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
log.print("/system/kernel: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Publish the loader's framebuffer as a claimable `display` device, so a
|
||||||
|
// user-space display service can take it over the same claim + mmio_map path as
|
||||||
|
// any other hardware (it is not firmware-discovered; it rides the boot handoff).
|
||||||
|
if (devices_broker.seedDisplay(fb.base, fb.width, fb.height, fb.pitch, @intFromEnum(fb.format))) |display_id| {
|
||||||
|
log.print("/system/kernel: framebuffer device {d} seeded ({d}x{d}, pitch {d}, write-combining)\n", .{ display_id, fb.width, fb.height, fb.pitch });
|
||||||
|
}
|
||||||
|
|
||||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||||
irq.init();
|
irq.init();
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||||
|
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||||
const pw = platform.powerInformation();
|
const pw = platform.powerInformation();
|
||||||
log.write("/system/kernel: power\n");
|
log.write("/system/kernel: power\n");
|
||||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||||
if (pw.s5) |s| {
|
|
||||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
|
||||||
} else {
|
|
||||||
log.write(" S5 slp_typ : (not found)\n");
|
|
||||||
}
|
|
||||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||||
|
|
||||||
// AML namespace parse integrity: consumed should equal total.
|
|
||||||
const am = platform.amlStats();
|
|
||||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
|
||||||
|
|
||||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||||
@@ -392,11 +416,22 @@ fn bringUpSecondaries() void {
|
|||||||
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
/// A user-facing status line. Now that the user-space **display service** owns the
|
||||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
/// framebuffer in normal operation (docs/display.md), routine kernel output goes to the
|
||||||
/// touches the framebuffer.
|
/// diagnostic `log` (serial/debugcon/RAM) *only* — never to the on-screen console, which
|
||||||
|
/// the compositor is about to paint over. For a message that must reach the screen even so
|
||||||
|
/// — a panic or a fatal fault, when the machine is going down — use `fatal`.
|
||||||
fn status(message: []const u8) void {
|
fn status(message: []const u8) void {
|
||||||
log.write(message);
|
log.write(message);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A fatal, user-facing message: to the diagnostic log *and* the on-screen console, forcing
|
||||||
|
/// the console back on (`setSuppressed(false)`) first — a dying machine's last words outrank
|
||||||
|
/// any display service holding the framebuffer. The console is otherwise silent in normal
|
||||||
|
/// operation (see `status`); it exists now only for early-boot and fatal output.
|
||||||
|
fn fatal(message: []const u8) void {
|
||||||
|
log.write(message);
|
||||||
|
console.setSuppressed(false);
|
||||||
console.write(message);
|
console.write(message);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -405,6 +440,11 @@ fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
|||||||
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn fatalPrint(comptime fmt: []const u8, args: anytype) void {
|
||||||
|
var buffer: [256]u8 = undefined;
|
||||||
|
fatal(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
/// Frames (4 KiB pages) to whole MiB.
|
/// Frames (4 KiB pages) to whole MiB.
|
||||||
fn mib(pages: u64) u64 {
|
fn mib(pages: u64) u64 {
|
||||||
return pages * abi.page_size / (1024 * 1024);
|
return pages * abi.page_size / (1024 * 1024);
|
||||||
@@ -465,16 +505,25 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
|||||||
|
|
||||||
log.checkpoint(cp_exception);
|
log.checkpoint(cp_exception);
|
||||||
const core = scheduler.currentCpuIndex();
|
const core = scheduler.currentCpuIndex();
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
// The machine is going down: paint the exception on screen too — `fatalPrint` forces the
|
||||||
// top of the diagnostic log.
|
// console back on even if a display service was holding the framebuffer — on top of the
|
||||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
// diagnostic log.
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
// Name the culprit: which task, and whether it faulted in ring 3 (a process the
|
||||||
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
// kernel would normally kill — landing here means it had no address space) or ring 0
|
||||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
// (the trusted base itself). Without this the fatal report is anonymous.
|
||||||
|
fatalPrint(" task : {d} ({s}), {s}\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe(), if (architecture.fromUser(state)) "ring 3 (user)" else "ring 0 (kernel)" });
|
||||||
|
fatalPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
|
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||||
|
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||||
|
if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||||
|
|
||||||
var buffer: [128]u8 = undefined;
|
var buffer: [128]u8 = undefined;
|
||||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||||
|
// Free the BKL if this core held it (a kernel-mode fault, or a nested fault in the
|
||||||
|
// recovery teardown), so halting this one core doesn't deadlock every other core on
|
||||||
|
// the lock. Only that core stops; the rest — and the supervisor — keep running.
|
||||||
|
sync.releaseIfHeldHere();
|
||||||
architecture.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -486,9 +535,11 @@ pub const panic = std.debug.FullPanic(struct {
|
|||||||
_ = first_trace_address;
|
_ = first_trace_address;
|
||||||
log.checkpoint(cp_panic);
|
log.checkpoint(cp_panic);
|
||||||
log.recordPanic(message);
|
log.recordPanic(message);
|
||||||
status("\nKERNEL PANIC: ");
|
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||||
status(message);
|
fatal(message);
|
||||||
status("\n");
|
fatal("\n");
|
||||||
|
fatalPrint(" task : {d} ({s})\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe() });
|
||||||
|
sync.releaseIfHeldHere(); // don't deadlock the other cores on the lock we may hold
|
||||||
architecture.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
}.panic);
|
}.panic);
|
||||||
|
|||||||
+131
-19
@@ -28,6 +28,7 @@ const parameters = @import("parameters");
|
|||||||
const architecture = @import("architecture");
|
const architecture = @import("architecture");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
|
const console = @import("console.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const ipc = @import("ipc-synchronous.zig");
|
const ipc = @import("ipc-synchronous.zig");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
@@ -80,9 +81,23 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
|||||||
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
|
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
|
||||||
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
|
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
|
||||||
|
|
||||||
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
/// The shared-memory arena: where `shm_create`/`shm_map` place shared cacheable regions, in
|
||||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
/// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by
|
||||||
const maximum_mmap_pages = 256;
|
/// a refcounted shm object and freed when its last capability drops, not on teardown, so the
|
||||||
|
/// mapping carries `device_grant`. Per-process cursor in `Task.shm_map_next` (docs/display-v2.md).
|
||||||
|
pub const shm_arena_base: u64 = 0x0000_7300_0000_0000;
|
||||||
|
pub const shm_arena_end: u64 = shm_arena_base + (256 << 20); // 256 MiB per process
|
||||||
|
|
||||||
|
/// Largest single `shm_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
|
||||||
|
/// also an overflow guard on the page count. shm frames are contiguous (like DMA), so this
|
||||||
|
/// bounds the contiguous allocation asked of the frame allocator.
|
||||||
|
const maximum_shm_pages = 8192;
|
||||||
|
|
||||||
|
/// Largest single `mmap` grant, in pages (32 MiB). Big enough for a display service's
|
||||||
|
/// back buffer at up to 4K (3840x2160x4 ≈ 8100 pages); the user heap otherwise grows in
|
||||||
|
/// small chunks. `systemMmap` maps page by page with rollback, so this is only a sanity
|
||||||
|
/// bound (and an overflow guard on the page count), not the size of any scratch array.
|
||||||
|
const maximum_mmap_pages = 8192;
|
||||||
|
|
||||||
/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn
|
/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn
|
||||||
/// parameters ("you are the driver for device 12"), not bulk data — IPC carries
|
/// parameters ("you are the driver for device 12"), not bulk data — IPC carries
|
||||||
@@ -209,6 +224,9 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
.timer_bind => systemTimerBind(state),
|
.timer_bind => systemTimerBind(state),
|
||||||
.klog_read => systemKlogRead(state),
|
.klog_read => systemKlogRead(state),
|
||||||
.wall_clock => systemWallClock(state),
|
.wall_clock => systemWallClock(state),
|
||||||
|
.shm_create => systemShmCreate(state),
|
||||||
|
.shm_map => systemShmMap(state),
|
||||||
|
.shm_physical => systemShmPhysical(state),
|
||||||
_ => fail(state),
|
_ => fail(state),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -301,12 +319,19 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||||
|
const device_id = architecture.systemCallArg(state, 0);
|
||||||
const claim_flags = sync.enter();
|
const claim_flags = sync.enter();
|
||||||
defer sync.leave(claim_flags);
|
defer sync.leave(claim_flags);
|
||||||
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
if (devices_broker.claim(device_id, scheduler.current().id)) {
|
||||||
architecture.setSystemCallResult(state, 0)
|
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||||
else
|
// so the kernel and the service don't scribble over each other's pixels. The
|
||||||
fail(state);
|
// claim releases (and the console resumes) automatically if the service dies;
|
||||||
|
// see releaseTaskResourcesLocked.
|
||||||
|
if (devices_broker.displayDevice()) |display_id| {
|
||||||
|
if (device_id == display_id) console.setSuppressed(true);
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
} else fail(state);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||||
@@ -341,7 +366,10 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
|||||||
const base_v = t.device_map_next;
|
const base_v = t.device_map_next;
|
||||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||||
|
|
||||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
|
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
|
||||||
|
// than the strong-uncacheable default that register MMIO needs.
|
||||||
|
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
|
||||||
|
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
|
||||||
t.device_map_next = base_v + pages * page_size;
|
t.device_map_next = base_v + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||||
}
|
}
|
||||||
@@ -447,6 +475,81 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||||
|
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
||||||
|
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
||||||
|
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
||||||
|
/// its frames are owned by a refcounted object: the handle is passed to another process as
|
||||||
|
/// an `ipc_call` send_cap, that process `shm_map`s it, and the frames free only when the
|
||||||
|
/// last capability drops (docs/display-v2.md — the compositor↔native-driver and
|
||||||
|
/// app↔compositor surface path).
|
||||||
|
fn systemShmCreate(state: *architecture.CpuState) void {
|
||||||
|
const len = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0 or len == 0) return fail(state);
|
||||||
|
|
||||||
|
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||||
|
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
||||||
|
|
||||||
|
// Reserve arena virtual space up front, so a mapping failure needs no rollback.
|
||||||
|
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||||
|
const base_v = t.shm_map_next;
|
||||||
|
if (base_v + pages * page_size > shm_arena_end) return fail(state); // arena exhausted
|
||||||
|
|
||||||
|
const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state);
|
||||||
|
// Zero through the physmap (the frames aren't mapped in the caller yet).
|
||||||
|
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||||
|
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||||
|
|
||||||
|
const shm = ipc.createShm(phys, pages) orelse {
|
||||||
|
for (0..pages) |i| pmm.free(phys + i * page_size);
|
||||||
|
return fail(state);
|
||||||
|
};
|
||||||
|
const handle = ipc.installShmHandle(t, shm);
|
||||||
|
if (handle < 0) {
|
||||||
|
ipc.dropShmRef(shm); // last ref: frees the object and its frames
|
||||||
|
return fail(state);
|
||||||
|
}
|
||||||
|
|
||||||
|
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
|
||||||
|
t.shm_map_next = base_v + pages * page_size;
|
||||||
|
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
|
||||||
|
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
|
||||||
|
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
||||||
|
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
||||||
|
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
||||||
|
fn systemShmMap(state: *architecture.CpuState) void {
|
||||||
|
const cap = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
|
||||||
|
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||||
|
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||||
|
const base_v = t.shm_map_next;
|
||||||
|
const size = shm.pages * page_size;
|
||||||
|
if (base_v + size > shm_arena_end) return fail(state);
|
||||||
|
|
||||||
|
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
|
||||||
|
t.shm_map_next = base_v + size;
|
||||||
|
architecture.setSystemCallResult(state, base_v);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a
|
||||||
|
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
||||||
|
/// physical base + length describes the whole region — which is exactly what a driver needs
|
||||||
|
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||||
|
/// capability can ask; there is no ambient way to turn a virtual address into a physical one.
|
||||||
|
fn systemShmPhysical(state: *architecture.CpuState) void {
|
||||||
|
const cap = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||||
|
architecture.setSystemCallResult(state, shm.phys);
|
||||||
|
}
|
||||||
|
|
||||||
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||||
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
||||||
/// enumerates it and hands each device it finds to the table, where a class driver
|
/// enumerates it and hands each device it finds to the table, where a class driver
|
||||||
@@ -601,6 +704,9 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
|||||||
recordExitLocked(t);
|
recordExitLocked(t);
|
||||||
irq.releaseOwner(t.id);
|
irq.releaseOwner(t.id);
|
||||||
devices_broker.releaseAllOwnedBy(t.id);
|
devices_broker.releaseAllOwnedBy(t.id);
|
||||||
|
// If that dropped the framebuffer claim (this task was the display service), let the
|
||||||
|
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
|
||||||
|
if (!devices_broker.displayClaimed()) console.setSuppressed(false);
|
||||||
// The dying task's signal endpoint and one-shot timers go with it.
|
// The dying task's signal endpoint and one-shot timers go with it.
|
||||||
if (t.signal_endpoint) |raw| {
|
if (t.signal_endpoint) |raw| {
|
||||||
ipc.dropRef(@ptrCast(@alignCast(raw)));
|
ipc.dropRef(@ptrCast(@alignCast(raw)));
|
||||||
@@ -631,6 +737,7 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
|||||||
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||||
}
|
}
|
||||||
ipc.abandonSenderLocked(t);
|
ipc.abandonSenderLocked(t);
|
||||||
|
ipc.killOwnedEndpointsLocked(t.id); // its registered services are gone: callers get -EPEER, not a hang
|
||||||
scheduler.removeFromWaitQueueLocked(t);
|
scheduler.removeFromWaitQueueLocked(t);
|
||||||
scheduler.forgetIpcClientLocked(t);
|
scheduler.forgetIpcClientLocked(t);
|
||||||
ipc.closeHandles(t);
|
ipc.closeHandles(t);
|
||||||
@@ -1024,21 +1131,26 @@ fn systemMmap(state: *architecture.CpuState) void {
|
|||||||
const base = t.heap_next;
|
const base = t.heap_next;
|
||||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||||
|
|
||||||
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
// Map page by page. On mid-way frame exhaustion, roll back the pages already mapped
|
||||||
// (no partially-mapped grant leaks into the address space).
|
// (unmap + free) so no partial grant leaks into the address space — the same
|
||||||
var frames: [maximum_mmap_pages]u64 = undefined;
|
// all-or-nothing guarantee as before, but without a fixed scratch array, so the
|
||||||
var got: usize = 0;
|
// per-call size can be a multi-MiB framebuffer.
|
||||||
while (got < pages) : (got += 1) {
|
var mapped: usize = 0;
|
||||||
frames[got] = pmm.alloc() orelse {
|
while (mapped < pages) : (mapped += 1) {
|
||||||
for (frames[0..got]) |f| pmm.free(f);
|
const frame = pmm.alloc() orelse {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < mapped) : (i += 1) {
|
||||||
|
const va = base + i * page_size;
|
||||||
|
if (architecture.translate(t.aspace, va)) |physical| {
|
||||||
|
architecture.unmapUserPageInto(t.aspace, va);
|
||||||
|
pmm.free(physical);
|
||||||
|
}
|
||||||
|
}
|
||||||
return fail(state);
|
return fail(state);
|
||||||
};
|
};
|
||||||
}
|
|
||||||
|
|
||||||
for (frames[0..pages], 0..) |frame, i| {
|
|
||||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||||
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
|
||||||
}
|
}
|
||||||
t.heap_next = base + pages * page_size;
|
t.heap_next = base + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base);
|
architecture.setSystemCallResult(state, base);
|
||||||
|
|||||||
@@ -86,9 +86,11 @@ pub const Task = struct {
|
|||||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||||
device_map_next: u64 = 0,
|
device_map_next: u64 = 0,
|
||||||
// --- synchronous IPC (ipc_sync.zig) ---
|
// --- synchronous IPC (ipc_sync.zig) ---
|
||||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
// Per-process handle table: a small-int handle names a kernel capability object.
|
||||||
// opaque here so the scheduler and IPC modules don't import each other.
|
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
||||||
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
// close/exit and cap-passing paths reclaim the right type. Kept opaque here so the
|
||||||
|
// scheduler and IPC modules don't import each other (ipc_sync.zig owns the kinds).
|
||||||
|
handles: [ipc_maximum_handles]?HandleObject = .{null} ** ipc_maximum_handles,
|
||||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||||
@@ -99,6 +101,7 @@ pub const Task = struct {
|
|||||||
ipc_reply_cap: u64 = 0,
|
ipc_reply_cap: u64 = 0,
|
||||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||||
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
||||||
|
shm_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
|
||||||
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
||||||
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
||||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||||
@@ -125,6 +128,13 @@ pub const maximum_task_name = abi.maximum_process_name;
|
|||||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||||
pub const ipc_maximum_handles = 16;
|
pub const ipc_maximum_handles = 16;
|
||||||
|
|
||||||
|
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||||
|
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||||
|
/// capability-passing path reclaim/share the right type. The `kind` values are defined by
|
||||||
|
/// ipc_sync.zig (`handle_kind_*`); kept an opaque `u8` here so the scheduler doesn't import
|
||||||
|
/// the IPC module.
|
||||||
|
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||||
var next_id: u32 = 1;
|
var next_id: u32 = 1;
|
||||||
|
|
||||||
@@ -791,6 +801,21 @@ pub fn currentCpuIndex() u32 {
|
|||||||
return thisCpu().index;
|
return thisCpu().index;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The running task's id, or 0 if this core's scheduler isn't up yet (early boot, no GS
|
||||||
|
/// base). Safe for a fault reporter to call unconditionally — like `currentCpuIndex`,
|
||||||
|
/// it never dereferences an unpublished per-CPU pointer and so can't fault a second time.
|
||||||
|
pub fn currentIdSafe() u32 {
|
||||||
|
if (architecture.cpuLocal() == 0) return 0;
|
||||||
|
return thisCpu().current.id;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The running task's name (argv[0]), or "" if this core's scheduler isn't up yet.
|
||||||
|
/// The companion to `currentIdSafe` for naming the culprit in a fatal fault report.
|
||||||
|
pub fn currentNameSafe() []const u8 {
|
||||||
|
if (architecture.cpuLocal() == 0) return "";
|
||||||
|
return thisCpu().current.name();
|
||||||
|
}
|
||||||
|
|
||||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||||
pub fn setPriority(p: Priority) void {
|
pub fn setPriority(p: Priority) void {
|
||||||
current().priority = p;
|
current().priority = p;
|
||||||
|
|||||||
@@ -33,6 +33,13 @@ const architecture = @import("architecture");
|
|||||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||||
var held = std.atomic.Value(u32).init(0);
|
var held = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
/// The per-CPU base pointer (`architecture.cpuLocal()`) of the core currently holding
|
||||||
|
/// the lock, or 0 when free. Metadata only — `held` is what enforces exclusion — read
|
||||||
|
/// solely by `releaseIfHeldHere` on the fatal-fault path. `cpuLocal()` is a unique,
|
||||||
|
/// architecture-level token per core (0 before this core's GS base is published, which
|
||||||
|
/// is fine: that window is single-core early boot, where no other core can deadlock).
|
||||||
|
var owner = std.atomic.Value(usize).init(0);
|
||||||
|
|
||||||
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
||||||
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
||||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||||
@@ -67,14 +74,29 @@ export fn releaseForFreshTask() callconv(.c) void {
|
|||||||
release();
|
release();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Release the big kernel lock **only if this core is the one holding it** — a no-op
|
||||||
|
/// otherwise. For the fatal-fault path (a kernel-mode fault, or a nested fault inside the
|
||||||
|
/// recovery teardown, both of which run under the lock): a core that dies holding the BKL
|
||||||
|
/// must free it, or every other core spins forever in `acquire` and the whole machine
|
||||||
|
/// deadlocks instead of just that core stopping. It must NOT free a lock another core
|
||||||
|
/// owns, hence the owner check. Caveat: if we held it mid-mutation the shared state may be
|
||||||
|
/// inconsistent — but letting the other cores (and the supervisor) run on possibly-degraded
|
||||||
|
/// state is strictly more recoverable than a guaranteed total hang.
|
||||||
|
pub fn releaseIfHeldHere() void {
|
||||||
|
const me = architecture.cpuLocal();
|
||||||
|
if (me != 0 and owner.load(.monotonic) == me) release();
|
||||||
|
}
|
||||||
|
|
||||||
fn acquire() void {
|
fn acquire() void {
|
||||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||||
while (held.swap(1, .acquire) != 0) {
|
while (held.swap(1, .acquire) != 0) {
|
||||||
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||||
}
|
}
|
||||||
|
owner.store(architecture.cpuLocal(), .monotonic);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn release() void {
|
fn release() void {
|
||||||
|
owner.store(0, .monotonic);
|
||||||
held.store(0, .release);
|
held.store(0, .release);
|
||||||
}
|
}
|
||||||
|
|||||||
+396
-103
@@ -95,6 +95,20 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
iommuTest();
|
iommuTest();
|
||||||
} else if (eql(case, "ioport")) {
|
} else if (eql(case, "ioport")) {
|
||||||
ioPortTest();
|
ioPortTest();
|
||||||
|
} else if (eql(case, "display")) {
|
||||||
|
displayTest(boot_information);
|
||||||
|
} else if (eql(case, "display-service")) {
|
||||||
|
displayServiceTest(boot_information);
|
||||||
|
} else if (eql(case, "display-demo")) {
|
||||||
|
displayDemoTest(boot_information);
|
||||||
|
} else if (eql(case, "shm")) {
|
||||||
|
shmTest(boot_information);
|
||||||
|
} else if (eql(case, "virtio-gpu")) {
|
||||||
|
virtioGpuTest(boot_information);
|
||||||
|
} else if (eql(case, "display-native")) {
|
||||||
|
displayNativeTest(boot_information);
|
||||||
|
} else if (eql(case, "display-reattach")) {
|
||||||
|
displayReattachTest(boot_information);
|
||||||
} else if (eql(case, "clock")) {
|
} else if (eql(case, "clock")) {
|
||||||
clockTest();
|
clockTest();
|
||||||
} else if (eql(case, "smp")) {
|
} else if (eql(case, "smp")) {
|
||||||
@@ -181,10 +195,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
containmentTest();
|
containmentTest();
|
||||||
} else if (eql(case, "device-manager")) {
|
} else if (eql(case, "device-manager")) {
|
||||||
deviceManagerTest(boot_information);
|
deviceManagerTest(boot_information);
|
||||||
} else if (eql(case, "poweroff")) {
|
|
||||||
powerTest(.off);
|
|
||||||
} else if (eql(case, "reboot")) {
|
} else if (eql(case, "reboot")) {
|
||||||
powerTest(.reboot);
|
rebootTest();
|
||||||
} else {
|
} else {
|
||||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||||
}
|
}
|
||||||
@@ -201,15 +213,14 @@ fn platformHal() platform.Hal {
|
|||||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
||||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
||||||
/// transition failed and we emit a FAIL result.
|
/// transition failed and we emit a FAIL result.
|
||||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
// Soft-off (S5) is no longer a kernel operation — the ring-3 acpi service owns it
|
||||||
const name = if (action == .off) "poweroff" else "reboot";
|
// (exercised end-to-end by `orderly-shutdown`). Reboot stays in the kernel (FADT
|
||||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
// reset register, no AML), so it keeps its own case.
|
||||||
|
fn rebootTest() void {
|
||||||
|
log("DANOS-TEST-BEGIN: reboot\n", .{});
|
||||||
const hal = platformHal();
|
const hal = platformHal();
|
||||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
log("DANOS-POWER: attempting reboot\n", .{});
|
||||||
switch (action) {
|
platform.reboot(hal);
|
||||||
.off => platform.shutdown(hal),
|
|
||||||
.reboot => platform.reboot(hal),
|
|
||||||
}
|
|
||||||
check("power transition took effect", false);
|
check("power transition took effect", false);
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -227,6 +238,19 @@ fn bufferHas(needle: []const u8) bool {
|
|||||||
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// How many `pci_device` functions the devices broker currently holds. A durable
|
||||||
|
/// snapshot, unlike a `bufferHas` poll of the single-latest write_buffer line, so a
|
||||||
|
/// test can wait on it without racing transient log output. `scratch` is
|
||||||
|
/// caller-owned to keep the (large) descriptor array off this helper's own frame.
|
||||||
|
fn brokerPciCount(scratch: []device_abi.DeviceDescriptor) u32 {
|
||||||
|
const k = @min(devices_broker.enumerate(scratch), scratch.len);
|
||||||
|
var count: u32 = 0;
|
||||||
|
for (scratch[0..k]) |d| {
|
||||||
|
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) count += 1;
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
/// Non-destructive checks of the memory map and frame allocator.
|
/// Non-destructive checks of the memory map and frame allocator.
|
||||||
fn smoke(boot_information: *const BootInformation) void {
|
fn smoke(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
||||||
@@ -276,22 +300,19 @@ fn timer() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Verify device discovery populated the platform facts the rest of the kernel
|
/// Verify device discovery populated the platform facts the rest of the kernel
|
||||||
/// depends on — the results ACPI parsing stashed in globals at boot. These are
|
/// depends on — the results the static ACPI tables stashed in globals at boot.
|
||||||
/// stable for the QEMU q35 + OVMF machine the harness runs, and span the tables:
|
/// These are stable for the QEMU q35 + OVMF machine the harness runs, and span the
|
||||||
/// MADT (LAPIC base, CPU count), FADT (PM/reset registers), and the AML parse
|
/// tables: MADT (LAPIC base, CPU count) and FADT (PM/reset registers). The kernel
|
||||||
/// (the sleep type, plus the integrity check that every byte was consumed).
|
/// no longer interprets AML — sleep types are the ring-3 acpi service's concern.
|
||||||
fn discoveryTest() void {
|
fn discoveryTest() void {
|
||||||
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
||||||
const pinfo = platform.platformInformation();
|
const pinfo = platform.platformInformation();
|
||||||
const pw = platform.powerInformation();
|
const pw = platform.powerInformation();
|
||||||
const am = platform.amlStats();
|
|
||||||
|
|
||||||
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
||||||
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
||||||
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
||||||
check("reset register supported (FADT)", pw.reset_supported);
|
check("reset register supported (FADT)", pw.reset_supported);
|
||||||
check("S5 sleep type found (AML)", pw.s5 != null);
|
|
||||||
check("AML parsed completely (consumed == total)", am.total > 0 and am.consumed == am.total);
|
|
||||||
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
||||||
|
|
||||||
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
||||||
@@ -1895,18 +1916,14 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Post-flip (M19.3) ground truth: the kernel no longer enumerates PCI
|
// Post-flip (M19.3) ground truth: the kernel no longer enumerates PCI
|
||||||
// functions, so equivalence inverts — the broker's function count after
|
// functions, so equivalence inverts — every PCI function the broker holds was
|
||||||
// the scan must equal what the driver itself reported finding.
|
// put there by the ring-3 driver, so before the driver runs the broker holds
|
||||||
|
// none. One reusable descriptor buffer (each snapshot is ~20 KiB; three live
|
||||||
|
// at once would overflow the 64 KiB bootstrap stack this test runs on).
|
||||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
check("the kernel seeded no PCI functions (the walk retired)", brokerPciCount(&buffer) == 0);
|
||||||
var boot_pci: u32 = 0;
|
|
||||||
for (buffer[0..n]) |d| {
|
|
||||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) boot_pci += 1;
|
|
||||||
}
|
|
||||||
check("the kernel seeded no PCI functions (the walk retired)", boot_pci == 0);
|
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
process.write_count = 0;
|
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
@@ -1917,68 +1934,49 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
|||||||
}
|
}
|
||||||
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
||||||
|
|
||||||
// First scan: wait for the driver's count line and parse the number.
|
// The manager spawns pci-bus, which scans the ECAM window and registers every
|
||||||
const count_prefix = "pci-bus: ";
|
// function it finds, so the broker's PCI count climbs from zero and plateaus.
|
||||||
const count_suffix = " functions found";
|
// Wait for it to *settle*: latch N only once the count has held steady for a
|
||||||
var reported: u32 = 0;
|
// stretch, so a mid-scan sample can't latch a low N that the rest of the same
|
||||||
scheduler.setPriority(1);
|
// scan then appears to exceed. The count is monotonic (registrations only add;
|
||||||
|
// the table has no unregister) so the plateau is permanent — stability is
|
||||||
|
// reached the moment the scan finishes and holds indefinitely. Sleep between
|
||||||
|
// samples rather than busy-yield: the boot context outranks the drivers, and a
|
||||||
|
// busy spin would starve the very processes it waits on; a sleeping task is
|
||||||
|
// woken by the timer, so the drivers run in between.
|
||||||
|
const poll_ms = 5;
|
||||||
|
var registered: u32 = 0;
|
||||||
|
var steady: u32 = 0;
|
||||||
var deadline = architecture.millis() + 15000;
|
var deadline = architecture.millis() + 15000;
|
||||||
while (architecture.millis() < deadline and reported == 0) {
|
while (architecture.millis() < deadline and steady < 60) { // 60 * 5ms = 300ms steady
|
||||||
const line = process.write_buffer[0..process.write_len];
|
const now = brokerPciCount(&buffer);
|
||||||
if (std.mem.indexOf(u8, line, count_prefix)) |start| {
|
if (now != 0 and now == registered) steady += 1 else steady = 0;
|
||||||
if (std.mem.indexOf(u8, line, count_suffix)) |digits_end| {
|
registered = now;
|
||||||
reported = std.fmt.parseInt(u32, line[start + count_prefix.len .. digits_end], 10) catch 0;
|
scheduler.sleep(poll_ms);
|
||||||
}
|
|
||||||
}
|
|
||||||
scheduler.yield();
|
|
||||||
}
|
}
|
||||||
scheduler.setPriority(4);
|
check("the ring-3 scan registered its PCI functions in the broker", registered >= 1);
|
||||||
check("the ring-3 scan reported a function count", reported >= 1);
|
|
||||||
|
|
||||||
// Every reported function was registered: the broker holds exactly them.
|
// The restart drill — the manager kills pci-bus ~1 s after its scan, prunes its
|
||||||
var registered: [64]device_abi.DeviceDescriptor = undefined;
|
// own child tree, and respawns it to re-claim, re-scan, and re-register — is
|
||||||
const r = @min(devices_broker.enumerate(®istered), registered.len);
|
// asserted by the harness's ordered regex over the whole serial log, the way
|
||||||
var registered_pci: u32 = 0;
|
// every restart drill is (see usbReportTest / driverRestartTest): the manager's
|
||||||
for (registered[0..r]) |d| {
|
// kill/prune/respawn lines are transient and would race a write_buffer poll, and
|
||||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) registered_pci += 1;
|
// the *broker* count can't witness the restart at all — the table has no
|
||||||
|
// unregister and register is idempotent (devices-broker.zig), so the kill leaves
|
||||||
|
// the nodes in place and the respawn's re-registration dedupes against them.
|
||||||
|
//
|
||||||
|
// That idempotence is exactly this test's kernel-side claim: watch, across the
|
||||||
|
// whole drill, that the count never grows past N. A broken dedup would append
|
||||||
|
// the re-scanned functions as duplicates (N -> 2N), and with no unregister that
|
||||||
|
// overshoot would persist — so a single late sample would catch it; the loop is
|
||||||
|
// belt-and-braces over the ~2 s the kill + backoff + respawn takes.
|
||||||
|
var duplicated = false;
|
||||||
|
deadline = architecture.millis() + 5000;
|
||||||
|
while (architecture.millis() < deadline and !duplicated) {
|
||||||
|
if (brokerPciCount(&buffer) > registered) duplicated = true;
|
||||||
|
scheduler.sleep(20);
|
||||||
}
|
}
|
||||||
check("the broker holds exactly the reported functions", registered_pci == reported);
|
check("no duplicate PCI nodes after the restart drill", !duplicated);
|
||||||
const kernel_count = reported; // the no-duplicate check below reuses it
|
|
||||||
|
|
||||||
// The restart drill: the manager kills pci-bus after its reports; the
|
|
||||||
// respawn re-claims, re-scans, and re-registers.
|
|
||||||
const restart_marker = "device-manager: restarting pci-bus";
|
|
||||||
scheduler.setPriority(1);
|
|
||||||
deadline = architecture.millis() + 15000;
|
|
||||||
var restarted = false;
|
|
||||||
while (architecture.millis() < deadline and !restarted) {
|
|
||||||
if (bufferHas(restart_marker)) restarted = true;
|
|
||||||
scheduler.yield();
|
|
||||||
}
|
|
||||||
scheduler.setPriority(4);
|
|
||||||
check("the manager restarted pci-bus", restarted);
|
|
||||||
|
|
||||||
var marker_buffer: [48]u8 = undefined;
|
|
||||||
const marker = std.fmt.bufPrint(&marker_buffer, "pci-bus: {d} functions found", .{reported}) catch "";
|
|
||||||
scheduler.setPriority(1);
|
|
||||||
deadline = architecture.millis() + 15000;
|
|
||||||
var seen = false;
|
|
||||||
while (architecture.millis() < deadline and !seen) {
|
|
||||||
if (bufferHas(marker)) seen = true;
|
|
||||||
scheduler.yield();
|
|
||||||
}
|
|
||||||
scheduler.setPriority(4);
|
|
||||||
check("the respawned scan reported the same count", seen);
|
|
||||||
|
|
||||||
// No duplicates: the registrations deduped against the kernel's own nodes
|
|
||||||
// on the first pass, and against themselves on the second.
|
|
||||||
var after: [64]device_abi.DeviceDescriptor = undefined;
|
|
||||||
const m = @min(devices_broker.enumerate(&after), after.len);
|
|
||||||
var after_count: u32 = 0;
|
|
||||||
for (after[0..m]) |d| {
|
|
||||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) after_count += 1;
|
|
||||||
}
|
|
||||||
check("no duplicate PCI nodes after register + restart + re-register", after_count == kernel_count);
|
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2095,11 +2093,11 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// M20.1: the ring-3 AML parse agrees with the kernel's. The manager spawns
|
/// M20.1: the ring-3 AML parse works. The manager spawns the discovery service
|
||||||
/// the discovery service (the acpi build variant); it claims the acpi-tables
|
/// (the acpi build variant); it claims the acpi-tables node, maps the blobs,
|
||||||
/// node, maps the blobs, parses them, and logs its Device count — which must
|
/// parses them, and self-verifies it found at least a floor of Device objects,
|
||||||
/// equal what the kernel's own parse produced (the equivalence that licenses
|
/// printing "acpi-parse: ok". The kernel no longer parses AML, so there is no
|
||||||
/// retiring the kernel's device build in M20.3).
|
/// kernel count to compare against — the ring-3 parse is now the only one.
|
||||||
fn acpiParseTest(boot_information: *const BootInformation) void {
|
fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
@@ -2114,23 +2112,18 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
// The kernel's own count, from the namespace it already built for \_S5.
|
// Spawn the discovery service directly with a device-count *floor* as argv:
|
||||||
const kernel_devices = platform.amlDeviceCount();
|
// it parses the blobs in ring 3 and self-verifies it found at least that many
|
||||||
check("the kernel namespace has devices to compare against", kernel_devices >= 1);
|
// Device objects, printing "acpi-parse: ok". A floor of 1 just proves the
|
||||||
|
// parser ran and produced a namespace (the QEMU q35 DSDT has dozens). The
|
||||||
// Spawn the discovery service directly with that count as argv: it parses
|
// marker is deterministic — no racing the shared serial buffer.
|
||||||
// the same blobs in ring 3 and self-verifies, printing "acpi-parse: ok" iff
|
|
||||||
// the counts match. The harness's expect regex is that marker — deterministic,
|
|
||||||
// no racing the shared serial buffer.
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
var count_text: [16]u8 = undefined;
|
|
||||||
const count_arg = std.fmt.bufPrint(&count_text, "{d}", .{kernel_devices}) catch "0";
|
|
||||||
var spawned = false;
|
var spawned = false;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(item.name, "discovery")) continue;
|
if (!eql(item.name, "discovery")) continue;
|
||||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", count_arg }, scheduler.currentId(), null) catch 0;
|
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
|
||||||
spawned = true;
|
spawned = true;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -2316,6 +2309,237 @@ fn inputTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// D2 — the display service comes up. Spawn it from the initial_ramdisk; it claims the
|
||||||
|
/// framebuffer the kernel seeded (D1), maps it write-combining, allocates a cacheable
|
||||||
|
/// back buffer, and proves the double-buffer path by clearing that buffer and presenting
|
||||||
|
/// it. Its `display: online WxH` + `display: presented frame 0` heartbeats are the
|
||||||
|
/// markers — seeing them proves a user-space compositor took the framebuffer and pushed
|
||||||
|
/// a whole composed frame to the screen, without ever drawing straight to the LFB.
|
||||||
|
fn displayServiceTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: display-service\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spawn the compositor and hand it the core. Its own serial heartbeats — `display:
|
||||||
|
// online WxH` and `display: presented frame 0` — are what the harness matches (it
|
||||||
|
// reads serial directly, like the fault cases). We don't poll for them in-kernel: a
|
||||||
|
// single service that comes up and blocks doesn't reschedule this bring-up context
|
||||||
|
// (there is no other runnable task to bounce control back through), so the honest
|
||||||
|
// observation point is the service's output itself, not a check() proxy here.
|
||||||
|
if (!spawnNamed(rd, "display")) {
|
||||||
|
log("display-service: could not spawn the display service\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
scheduler.setPriority(1); // below the service, so it runs and comes up first
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// D4 — a separate process drives the compositor. Spawn the display service and the
|
||||||
|
/// hardware-free `display-demo` client, which creates a wallpaper, a moving rectangle,
|
||||||
|
/// and a cursor and presents a run of frames. Its `display-demo: ok` heartbeat — printed
|
||||||
|
/// only after it drove frames of motion through the layer client API and the compositor —
|
||||||
|
/// is the harness's marker (the visible motion itself is a screenshot away via run-x86-64).
|
||||||
|
/// The demo keeps presenting, so unlike a lone blocking service the scheduler stays busy;
|
||||||
|
/// we still match on serial rather than poll, for consistency.
|
||||||
|
fn displayDemoTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: display-demo\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!spawnNamed(rd, "display")) {
|
||||||
|
log("display-demo: could not spawn the display service\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = spawnNamed(rd, "display-demo");
|
||||||
|
scheduler.setPriority(1); // below the service + demo, so they run
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// V2 — cross-process shared memory (docs/display-v2.md). Spawn shm-server and shm-client:
|
||||||
|
/// the client shm_creates a region, writes a pattern, and passes the region's capability to
|
||||||
|
/// the server as an ipc_call send_cap; the server shm_maps it and confirms the pattern is
|
||||||
|
/// visible — proving the two processes share the same physical pages, and that the extended
|
||||||
|
/// capability-passing (endpoints → memory objects) works. Its `shm: shared 4096 bytes ok`
|
||||||
|
/// heartbeat is the marker.
|
||||||
|
fn shmTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: shm\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!spawnNamed(rd, "shm-server")) {
|
||||||
|
log("shm: could not spawn shm-server\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = spawnNamed(rd, "shm-client");
|
||||||
|
scheduler.setPriority(1); // below the two, so they run
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// V3 — the virtio-gpu driver, end to end (docs/display-v2.md). Boot the device-manager
|
||||||
|
/// stack (in its normal mode) so it discovers the virtio-gpu PCI function — present because
|
||||||
|
/// the harness boots this case with QEMU's `-device virtio-gpu-pci` — and spawns the driver.
|
||||||
|
/// The driver claims the device, brings up the control virtqueue, creates a 2D scanout,
|
||||||
|
/// paints a known pattern, flushes it, and waits for the device's used-ring ack. Its serial
|
||||||
|
/// heartbeats — `virtio-gpu: scanout WxH online` and `virtio-gpu: flush acked, pixel check
|
||||||
|
/// ok` — are the harness's markers (it reads serial directly, like the display cases). The
|
||||||
|
/// used-ring ack is the device confirming it consumed the frame; the pixel read-back proves
|
||||||
|
/// the backing is CPU-visible RAM — together the automated stand-in for "it's on screen".
|
||||||
|
fn virtioGpuTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: virtio-gpu\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spawn device-manager in its normal mode: its initialise discovers the PCI host bridge
|
||||||
|
// from the kernel device tree, spawns pci-bus, and matches the virtio-gpu class triple to
|
||||||
|
// spawn our driver with the function's device id as argv[1].
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (manager == 0) {
|
||||||
|
log("virtio-gpu: could not spawn device-manager\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
scheduler.setPriority(1); // below the manager and the driver it spawns, so they run
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// V4 — the native backend + hot-attach (docs/display-v2.md). Boot the compositor and the
|
||||||
|
/// hardware-free `display-demo` client (as displayDemoTest does), then the device-manager
|
||||||
|
/// stack so it discovers the virtio-gpu function — present via QEMU's `-device
|
||||||
|
/// virtio-gpu-pci` — and spawns the driver. The driver brings up its scanout, then announces
|
||||||
|
/// the shared surface to the already-running compositor, which maps it, upgrades off the GOP
|
||||||
|
/// floor, and presents the composited frame through the native backend. Its serial heartbeats
|
||||||
|
/// — `display: scanout upgraded to virtio-gpu` and `display: native present verified` — plus
|
||||||
|
/// the demo's own `display-demo: ok` are the harness's markers. Display is spawned first so
|
||||||
|
/// it is registered on `.display` before the driver announces.
|
||||||
|
fn displayNativeTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: display-native\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (manager == 0) {
|
||||||
|
log("display-native: could not spawn device-manager\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!spawnNamed(rd, "display")) {
|
||||||
|
log("display-native: could not spawn the display service\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = spawnNamed(rd, "display-demo");
|
||||||
|
scheduler.setPriority(1); // below the compositor, the demo, and the driver, so they run
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// V6 — resilience: the compositor survives the virtio-gpu driver dying and re-attaches when
|
||||||
|
/// device-manager restarts it (docs/display-v2.md). Same boot as display-native, but the
|
||||||
|
/// manager runs in "test-scanout-restart" mode: a moment after the driver hellos, it kills it
|
||||||
|
/// once; the normal restart policy respawns it, the restarted driver re-announces, and the
|
||||||
|
/// compositor re-attaches to the fresh scanout — logging `display: scanout re-attached` after
|
||||||
|
/// the initial `display: scanout upgraded to virtio-gpu`. The compositor must not crash.
|
||||||
|
fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: display-reattach\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (manager == 0) {
|
||||||
|
log("display-reattach: could not spawn device-manager\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!spawnNamed(rd, "display")) {
|
||||||
|
log("display-reattach: could not spawn the display service\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = spawnNamed(rd, "display-demo");
|
||||||
|
scheduler.setPriority(1); // below the compositor, the demo, and the driver, so they run
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||||
@@ -2638,8 +2862,8 @@ fn ioPassTest() void {
|
|||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
// Map it the way mmio_map does (device grant), then tear the space down.
|
// Map it the way mmio_map does (device grant, strong-uncacheable), then tear the space down.
|
||||||
architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, abi.page_size);
|
architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, abi.page_size, false);
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(aspace);
|
||||||
|
|
||||||
// The page tables were reclaimed; the device-granted frame must not have been.
|
// The page tables were reclaimed; the device-granted frame must not have been.
|
||||||
@@ -2649,6 +2873,75 @@ fn ioPassTest() void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// D1 — the framebuffer handoff primitive. The kernel seeds the loader's framebuffer as
|
||||||
|
/// a claimable `display` device with a write-combining `memory` resource; a display
|
||||||
|
/// service reaches it over the ordinary claim + mmio_map path. Prove the whole chain:
|
||||||
|
/// the node is present and correctly shaped, it maps, and — the point of D1 — the
|
||||||
|
/// mapping is genuinely write-combining, not the strong-uncacheable default that would
|
||||||
|
/// make a framebuffer blit glacial.
|
||||||
|
fn displayTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: display\n", .{});
|
||||||
|
const fb = boot_information.framebuffer;
|
||||||
|
if (!fb.present()) {
|
||||||
|
// Headless: nothing to seed. Not a failure of the mechanism, so pass cleanly.
|
||||||
|
log("display: no framebuffer (headless); skipping\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The kernel seeded a display device in kmain, right after devices_broker.init.
|
||||||
|
const display_id = devices_broker.displayDevice() orelse {
|
||||||
|
check("a framebuffer display device was seeded", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||||
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||||
|
check("the seeded display id is enumerable", display_id < n);
|
||||||
|
if (display_id >= n) {
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const d = buffer[@intCast(display_id)];
|
||||||
|
|
||||||
|
check("the node is class display", d.class == @intFromEnum(device_abi.DeviceClass.display));
|
||||||
|
check("it carries the framebuffer geometry", d.display.width == fb.width and d.display.height == fb.height and d.display.pitch == fb.pitch);
|
||||||
|
check("it has exactly one resource", d.resource_count == 1);
|
||||||
|
const r = d.resources[0];
|
||||||
|
check("that resource is a memory window", r.kind == @intFromEnum(device_abi.ResourceKind.memory));
|
||||||
|
check("it spans the whole framebuffer", r.start == fb.base and r.len == @as(u64, fb.height) * fb.pitch);
|
||||||
|
check("it is flagged write-combining", (r.flags & device_abi.resource_flag_write_combining) != 0);
|
||||||
|
|
||||||
|
// Walk the real claim + map path a display service would, into a throwaway address
|
||||||
|
// space, and confirm the leaf's cache type. We never run this space (no CR3 load) —
|
||||||
|
// we only read back the page-table entries — so aliasing the same physical page at
|
||||||
|
// two cache types below is inert.
|
||||||
|
const aspace = architecture.createAddressSpace() orelse {
|
||||||
|
check("created a fresh address space", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
defer architecture.destroyAddressSpace(aspace);
|
||||||
|
|
||||||
|
const page_base = fb.base & ~@as(u64, abi.page_size - 1);
|
||||||
|
architecture.mapUserDeviceInto(aspace, process.device_arena_base, page_base, abi.page_size, true);
|
||||||
|
check(
|
||||||
|
"the framebuffer maps write-combining (PAT entry 4: PAT bit set, PCD/PWT clear)",
|
||||||
|
architecture.userLeafIsWriteCombining(aspace, process.device_arena_base) == true,
|
||||||
|
);
|
||||||
|
|
||||||
|
// Regression guard: the strong-uncacheable default is still that, so WC is a real
|
||||||
|
// choice the flag makes, not the only behaviour.
|
||||||
|
architecture.mapUserDeviceInto(aspace, process.device_arena_base + abi.page_size, page_base, abi.page_size, false);
|
||||||
|
check(
|
||||||
|
"a register window still maps strong-uncacheable",
|
||||||
|
architecture.userLeafIsWriteCombining(aspace, process.device_arena_base + abi.page_size) == false,
|
||||||
|
);
|
||||||
|
|
||||||
|
log("display: mapped {d}x{d} pitch {d} (write-combining)\n", .{ fb.width, fb.height, fb.pitch });
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
fn faultInvalidOpcode() void {
|
fn faultInvalidOpcode() void {
|
||||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
||||||
asm volatile ("ud2");
|
asm volatile ("ud2");
|
||||||
|
|||||||
@@ -103,9 +103,11 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||||
// own device count to self-verify against — deterministic, no log-scraping.
|
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||||
|
// devices is the check. Deterministic, no log-scraping.
|
||||||
|
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||||
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||||
@@ -162,11 +164,11 @@ pub fn main(init: runtime.process.Init) void {
|
|||||||
var namespace = result.namespace;
|
var namespace = result.namespace;
|
||||||
const devices = aml.deviceCount(&namespace);
|
const devices = aml.deviceCount(&namespace);
|
||||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||||
if (expected) |want| {
|
if (floor) |minimum| {
|
||||||
if (devices == want) {
|
if (devices >= minimum) {
|
||||||
_ = runtime.system.write("acpi-parse: ok\n");
|
_ = runtime.system.write("acpi-parse: ok\n");
|
||||||
} else {
|
} else {
|
||||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
writeLine("acpi-parse: too few (ring-3 {d} < floor {d})\n", .{ devices, minimum });
|
||||||
}
|
}
|
||||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||||
while (true) runtime.system.sleep(1000);
|
while (true) runtime.system.sleep(1000);
|
||||||
|
|||||||
@@ -16,6 +16,10 @@ pub const Operation = enum(u32) {
|
|||||||
read = 1,
|
read = 1,
|
||||||
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||||
write = 2,
|
write = 2,
|
||||||
|
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||||
|
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||||
|
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||||
|
flush = 3,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub const Request = extern struct {
|
pub const Request = extern struct {
|
||||||
|
|||||||
@@ -41,6 +41,15 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
|||||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||||
});
|
});
|
||||||
|
|
||||||
|
/// The PCI class triple of a virtio-gpu — Display Controller / Other (0x80) / 0. The class
|
||||||
|
/// alone cannot tell it from any other display/other function, so the driver re-confirms
|
||||||
|
/// vendor 0x1AF4 / device 0x1050 from config space once spawned; this only gets it spawned.
|
||||||
|
const virtio_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||||
|
.base = @intFromEnum(pci_class.BaseClass.display),
|
||||||
|
.subclass = 0x80, // "Other" — no named SubClass member (PCI convention)
|
||||||
|
.prog_if = 0,
|
||||||
|
});
|
||||||
|
|
||||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||||
/// several identical controllers — one driver instance per reported device,
|
/// several identical controllers — one driver instance per reported device,
|
||||||
@@ -48,6 +57,7 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
|||||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||||
return switch (identity) {
|
return switch (identity) {
|
||||||
xhci_pci_class => "usb-xhci-bus",
|
xhci_pci_class => "usb-xhci-bus",
|
||||||
|
virtio_gpu_pci_class => "virtio-gpu",
|
||||||
else => null,
|
else => null,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -149,6 +159,8 @@ var test_restart_mode = false;
|
|||||||
var test_usb_restart_mode = false;
|
var test_usb_restart_mode = false;
|
||||||
var test_usb_killed = false;
|
var test_usb_killed = false;
|
||||||
var test_pci_restart_mode = false;
|
var test_pci_restart_mode = false;
|
||||||
|
var test_scanout_restart_mode = false;
|
||||||
|
var test_scanout_killed = false;
|
||||||
var test_kill_pid: u32 = 0;
|
var test_kill_pid: u32 = 0;
|
||||||
var test_kill_due_ns: u64 = 0;
|
var test_kill_due_ns: u64 = 0;
|
||||||
|
|
||||||
@@ -415,6 +427,14 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
|||||||
} else if (driverByProcess(sender)) |driver| {
|
} else if (driverByProcess(sender)) |driver| {
|
||||||
driver.state = .running;
|
driver.state = .running;
|
||||||
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||||
|
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||||
|
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||||
|
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "virtio-gpu")) {
|
||||||
|
test_scanout_killed = true;
|
||||||
|
test_kill_pid = sender;
|
||||||
|
test_kill_due_ns = system.clock() + 1_500_000_000;
|
||||||
|
_ = system.timerOnce(manager_endpoint, 1600);
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
status = -1;
|
status = -1;
|
||||||
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||||
@@ -557,6 +577,7 @@ pub fn main(init: runtime.process.Init) void {
|
|||||||
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
||||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||||
|
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||||
}
|
}
|
||||||
runtime.service.run(protocol.message_maximum, .{
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
.service = .device_manager,
|
.service = .device_manager,
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
//! system/services/display-demo — a hardware-free client of the display service, the
|
||||||
|
//! `input-source` analog for the compositor. It creates a wallpaper, a rectangle it moves
|
||||||
|
//! each frame, and a small cursor, then drives the compositor in a present loop — proof
|
||||||
|
//! that a *separate process* can compose a moving scene through the display service over
|
||||||
|
//! IPC, exercising the layer client API and damage-driven present end to end
|
||||||
|
//! (docs/display.md). It logs `display-demo: ok` once it has driven a run of frames.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const display = runtime.display;
|
||||||
|
const system = runtime.system;
|
||||||
|
const time = runtime.time;
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
const mode = display.info() orelse {
|
||||||
|
_ = system.write("display-demo: no display service\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// A full-screen wallpaper under everything.
|
||||||
|
const wallpaper = display.createLayer(0, 0, mode.width, mode.height, 0) orelse return createFailed();
|
||||||
|
_ = wallpaper.fill(0, 0, mode.width, mode.height, display.color(0x10, 0x18, 0x28));
|
||||||
|
|
||||||
|
// A rectangle that slides back and forth.
|
||||||
|
const box_w: u32 = 140;
|
||||||
|
const box_h: u32 = 100;
|
||||||
|
const box_y: i32 = 200;
|
||||||
|
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
||||||
|
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
||||||
|
|
||||||
|
// A little cursor on top.
|
||||||
|
const cursor = display.createLayer(40, 40, 12, 12, 2) orelse return createFailed();
|
||||||
|
_ = cursor.fill(0, 0, 12, 12, display.color(0xF0, 0xF0, 0xF0));
|
||||||
|
|
||||||
|
_ = display.present();
|
||||||
|
_ = system.write("display-demo: scene up; animating\n");
|
||||||
|
|
||||||
|
const span: i32 = @as(i32, @intCast(mode.width)) - @as(i32, @intCast(box_w));
|
||||||
|
var x: i32 = 0;
|
||||||
|
var dx: i32 = 8;
|
||||||
|
var frame: u32 = 0;
|
||||||
|
while (true) : (frame += 1) {
|
||||||
|
x += dx;
|
||||||
|
if (x <= 0) {
|
||||||
|
x = 0;
|
||||||
|
dx = -dx;
|
||||||
|
} else if (x >= span) {
|
||||||
|
x = span;
|
||||||
|
dx = -dx;
|
||||||
|
}
|
||||||
|
_ = box.configure(x, box_y, 1, true); // move it; the compositor repaints old + new
|
||||||
|
_ = display.present();
|
||||||
|
// A run of frames drawn through the compositor is the automated proof (the visible
|
||||||
|
// motion is a screenshot away via `zig build run-x86-64`).
|
||||||
|
if (frame == 20) _ = system.write("display-demo: ok\n");
|
||||||
|
time.sleep(time.Duration.fromMillis(30));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn createFailed() void {
|
||||||
|
_ = system.write("display-demo: create failed\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,273 @@
|
|||||||
|
//! The compositor's **scanout backend** — how a finished frame reaches the panel
|
||||||
|
//! (docs/display-v2.md). The compositor composes its layer stack into the backend's
|
||||||
|
//! cacheable `surface()` and calls `present(damage)`; everything device-specific lives
|
||||||
|
//! here. Today there is one backend, `Gop` — the firmware framebuffer: a cacheable back
|
||||||
|
//! buffer streamed write-combining to the linear framebuffer. A native virtio-gpu backend
|
||||||
|
//! slots in beside it later (V4); the compositor never learns which is active.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const compositor = @import("compositor.zig");
|
||||||
|
|
||||||
|
const system = runtime.system;
|
||||||
|
const device = runtime.device;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const scanout_protocol = runtime.scanout_protocol;
|
||||||
|
const Rect = compositor.Rect;
|
||||||
|
const Surface = compositor.Surface;
|
||||||
|
|
||||||
|
/// The current display mode, as a backend reports it.
|
||||||
|
pub const Info = struct { width: u32, height: u32, pitch: u32, format: u32 };
|
||||||
|
|
||||||
|
/// Enumeration scratch — a `DeviceDescriptor` is large, and only one scan is ever needed.
|
||||||
|
var device_table: [64]device.DeviceDescriptor = undefined;
|
||||||
|
|
||||||
|
/// The GOP framebuffer backend: claims the kernel-seeded `display` device, maps the linear
|
||||||
|
/// framebuffer write-combining as the front buffer, and keeps a cacheable back buffer of
|
||||||
|
/// the same geometry as the compose target. `present` streams the damaged rectangle from
|
||||||
|
/// the back buffer to the LFB (sequential WC writes; the LFB is never read). No mode-set,
|
||||||
|
/// no vsync — the portable floor (docs/display-v2.md).
|
||||||
|
pub const Gop = struct {
|
||||||
|
device_id: u64,
|
||||||
|
front: [*]volatile u8, // the LFB (write-combining)
|
||||||
|
back: [*]u8, // cacheable compose target, same geometry
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
pitch: u32,
|
||||||
|
format: u32,
|
||||||
|
|
||||||
|
/// The framebuffer's id and geometry, captured together. `findDisplay` reads these out of
|
||||||
|
/// the enumeration table and returns them by value, so the caller never re-reads the table
|
||||||
|
/// across later syscalls (`device_enumerate` writes the whole table straight into this
|
||||||
|
/// process's memory; reading a descriptor's tail again after other syscalls have run is a
|
||||||
|
/// window we simply avoid by copying the few fields we need up front).
|
||||||
|
const Found = struct { id: u64, width: u32, height: u32, pitch: u32, format: u32 };
|
||||||
|
|
||||||
|
/// The first `display`-class device with a *valid* (non-zero) geometry, or null. A zero
|
||||||
|
/// geometry is treated as "not ready yet" so the caller retries — a real framebuffer always
|
||||||
|
/// has a non-zero width, height, and pitch.
|
||||||
|
fn findDisplay() ?Found {
|
||||||
|
const total = device.enumerate(&device_table);
|
||||||
|
const n = @min(total, device_table.len);
|
||||||
|
for (device_table[0..n]) |*d| {
|
||||||
|
if (d.class != @intFromEnum(device.DeviceClass.display)) continue;
|
||||||
|
if (d.display.width == 0 or d.display.height == 0 or d.display.pitch == 0) continue;
|
||||||
|
return .{ .id = d.id, .width = d.display.width, .height = d.display.height, .pitch = d.display.pitch, .format = d.display.format };
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Claim the framebuffer (retrying while discovery catches up), map the LFB, and
|
||||||
|
/// allocate the back buffer. Null if there is no framebuffer or a mapping fails.
|
||||||
|
pub fn init() ?Gop {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
const found = while (tries < 100) : (tries += 1) {
|
||||||
|
if (findDisplay()) |f| break f;
|
||||||
|
system.sleep(50);
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: no framebuffer device (headless?)\n");
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!device.claim(found.id)) {
|
||||||
|
_ = system.write("display: could not claim the framebuffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
// Resource 0 is the framebuffer memory window; the kernel maps it write-combining
|
||||||
|
// because the resource carries that flag (docs/display-plan.md D1).
|
||||||
|
const front_base = device.mmioMap(found.id, 0) orelse {
|
||||||
|
_ = system.write("display: could not map the framebuffer\n");
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
const size = @as(usize, found.height) * found.pitch;
|
||||||
|
const back_base = system.mmap(size, system.PROT_READ | system.PROT_WRITE);
|
||||||
|
if (system.mmapFailed(back_base)) {
|
||||||
|
_ = system.write("display: could not allocate the back buffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
return .{
|
||||||
|
.device_id = found.id,
|
||||||
|
.front = @ptrFromInt(front_base),
|
||||||
|
.back = @ptrFromInt(back_base),
|
||||||
|
.width = found.width,
|
||||||
|
.height = found.height,
|
||||||
|
.pitch = found.pitch,
|
||||||
|
.format = found.format,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn info(self: *const Gop) Info {
|
||||||
|
return .{ .width = self.width, .height = self.height, .pitch = self.pitch, .format = self.format };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cacheable compose target (the back buffer).
|
||||||
|
pub fn surface(self: *const Gop) Surface {
|
||||||
|
return .{
|
||||||
|
.pixels = @ptrCast(@alignCast(self.back)),
|
||||||
|
.stride = self.pitch / 4, // pitch is bytes; a 32-bpp row is pitch/4 pixels
|
||||||
|
.width = self.width,
|
||||||
|
.height = self.height,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stream the damaged rectangle from the back buffer to the write-combining LFB, row by
|
||||||
|
/// row (sequential writes — what WC memory wants; the LFB is never read).
|
||||||
|
pub fn present(self: *const Gop, damage: Rect) void {
|
||||||
|
const c = damage.intersect(.{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) });
|
||||||
|
if (c.isEmpty()) return;
|
||||||
|
var y: i32 = c.y;
|
||||||
|
while (y < c.bottom()) : (y += 1) {
|
||||||
|
const off = @as(usize, @intCast(y)) * self.pitch;
|
||||||
|
const src: [*]const u32 = @ptrCast(@alignCast(self.back + off));
|
||||||
|
const dst: [*]volatile u32 = @ptrCast(@alignCast(self.front + off));
|
||||||
|
var x: i32 = c.x;
|
||||||
|
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A display mode the native backend can switch to.
|
||||||
|
pub const Mode = scanout_protocol.Mode;
|
||||||
|
|
||||||
|
/// The native virtio-gpu backend: the compositor composes into a **shared** scanout surface
|
||||||
|
/// (an `shm` region the driver created and handed over) and `present` asks the driver to put
|
||||||
|
/// a frame on the panel over its `.scanout` endpoint. Unlike GOP there is no local copy — the
|
||||||
|
/// surface *is* the device's resource backing, so compositing writes land straight where the
|
||||||
|
/// driver transfers-and-flushes from (x86 DMA is cache-coherent, so the cacheable shared pages
|
||||||
|
/// need no explicit flush). Built by the display service when a driver announces (V4). The
|
||||||
|
/// surface is sized to the driver's largest mode, so `stride` (its row width) is fixed while
|
||||||
|
/// `width`/`height` — the active mode — change under `setMode` (V5).
|
||||||
|
pub const VirtioGpu = struct {
|
||||||
|
pixels: [*]u32, // the shared scanout surface, mapped into the compositor
|
||||||
|
stride: u32, // the surface's row stride in pixels (the driver's max mode width) — fixed
|
||||||
|
width: u32, // the active mode
|
||||||
|
height: u32,
|
||||||
|
format: u32,
|
||||||
|
scanout: ipc.Handle, // the driver's present + mode channel (looked up on `.scanout`)
|
||||||
|
|
||||||
|
pub fn info(self: *const VirtioGpu) Info {
|
||||||
|
return .{ .width = self.width, .height = self.height, .pitch = self.stride * 4, .format = self.format };
|
||||||
|
}
|
||||||
|
pub fn surface(self: *const VirtioGpu) Surface {
|
||||||
|
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
|
||||||
|
}
|
||||||
|
/// Ask the driver to present. The composited pixels are already in the shared surface, so
|
||||||
|
/// this is a single request over `.scanout`; the driver transfers + fenced-flushes.
|
||||||
|
pub fn present(self: *const VirtioGpu, damage: Rect) void {
|
||||||
|
_ = damage;
|
||||||
|
var request = scanout_protocol.Request{
|
||||||
|
.operation = @intFromEnum(scanout_protocol.Operation.present),
|
||||||
|
.width = self.width,
|
||||||
|
.height = self.height,
|
||||||
|
};
|
||||||
|
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||||
|
_ = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch {};
|
||||||
|
}
|
||||||
|
/// Fill `out` with the driver's offered modes; returns how many were written.
|
||||||
|
pub fn modes(self: *const VirtioGpu, out: []Mode) usize {
|
||||||
|
var request = scanout_protocol.Request{ .operation = @intFromEnum(scanout_protocol.Operation.get_modes) };
|
||||||
|
var reply: [scanout_protocol.modes_reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return 0;
|
||||||
|
if (n < scanout_protocol.modes_reply_size) return 0;
|
||||||
|
const answer = std.mem.bytesToValue(scanout_protocol.ModesReply, reply[0..scanout_protocol.modes_reply_size]);
|
||||||
|
if (answer.status != 0) return 0;
|
||||||
|
const count = @min(@min(answer.count, scanout_protocol.max_modes), out.len);
|
||||||
|
for (0..count) |i| out[i] = answer.modes[i];
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
/// Change the scanout resolution. On success the active `width`/`height` update (the shared
|
||||||
|
/// surface — sized to the max mode — is unchanged, so `stride` stays put).
|
||||||
|
pub fn setMode(self: *VirtioGpu, w: u32, h: u32) bool {
|
||||||
|
if (w == 0 or h == 0 or w > self.stride) return false;
|
||||||
|
var request = scanout_protocol.Request{
|
||||||
|
.operation = @intFromEnum(scanout_protocol.Operation.set_mode),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
};
|
||||||
|
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (n < scanout_protocol.reply_size) return false;
|
||||||
|
if (std.mem.bytesToValue(scanout_protocol.Reply, reply[0..scanout_protocol.reply_size]).status != 0) return false;
|
||||||
|
self.width = w;
|
||||||
|
self.height = h;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The pluggable scanout backend. A tagged union so the compositor holds one value and
|
||||||
|
/// dispatches without caring which is active; the `virtio` native backend joins `gop` at V4.
|
||||||
|
pub const Backend = union(enum) {
|
||||||
|
gop: Gop,
|
||||||
|
virtio: VirtioGpu,
|
||||||
|
|
||||||
|
pub fn info(self: *const Backend) Info {
|
||||||
|
return switch (self.*) {
|
||||||
|
inline else => |*b| b.info(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
pub fn surface(self: *const Backend) Surface {
|
||||||
|
return switch (self.*) {
|
||||||
|
inline else => |*b| b.surface(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
pub fn present(self: *const Backend, damage: Rect) void {
|
||||||
|
switch (self.*) {
|
||||||
|
inline else => |*b| b.present(damage),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// The modes this backend can switch to (none for GOP); returns how many were written.
|
||||||
|
pub fn modes(self: *const Backend, out: []Mode) usize {
|
||||||
|
return switch (self.*) {
|
||||||
|
.virtio => |*v| v.modes(out),
|
||||||
|
.gop => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
/// Change the resolution; false if this backend can't mode-set or the mode was refused.
|
||||||
|
pub fn setMode(self: *Backend, w: u32, h: u32) bool {
|
||||||
|
return switch (self.*) {
|
||||||
|
.virtio => |*v| v.setMode(w, h),
|
||||||
|
.gop => false,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
/// Whether this backend supports runtime mode-setting (GOP: no; virtio-gpu: yes, V5).
|
||||||
|
pub fn canModeSet(self: *const Backend) bool {
|
||||||
|
return switch (self.*) {
|
||||||
|
.gop => false,
|
||||||
|
.virtio => true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
/// Whether this backend has a vblank/fence for tear-free present (virtio-gpu: yes, V5 — every
|
||||||
|
/// flush is fenced, so the device signals completion when the frame is actually on screen).
|
||||||
|
pub fn hasVsync(self: *const Backend) bool {
|
||||||
|
return switch (self.*) {
|
||||||
|
.gop => false,
|
||||||
|
.virtio => true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Which backend to use. The pure selection *decision* is `chooseKind`; `select` below
|
||||||
|
/// binds it to the (syscall-bound) bring-up.
|
||||||
|
pub const Kind = enum { gop, virtio };
|
||||||
|
|
||||||
|
/// The selection decision, factored out of bring-up so it stays pure and host-testable:
|
||||||
|
/// prefer a native driver when one has announced itself (docs/display-v2.md V4), else the
|
||||||
|
/// GOP floor. Trivial today; it grows real inputs when native detection lands.
|
||||||
|
pub fn chooseKind(native_available: bool) Kind {
|
||||||
|
return if (native_available) .virtio else .gop;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pick and bring up the best available backend. Today the GOP framebuffer is the only one
|
||||||
|
/// (`chooseKind(false)` → `.gop`), so this is `Gop.init()`. V4 adds the native-if-present
|
||||||
|
/// branch, with GOP as the floor.
|
||||||
|
pub fn select() ?Backend {
|
||||||
|
return switch (chooseKind(false)) {
|
||||||
|
.gop => .{ .gop = Gop.init() orelse return null },
|
||||||
|
.virtio => unreachable, // no native detection yet (V4)
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "selection prefers native when present, else the gop floor" {
|
||||||
|
try std.testing.expectEqual(Kind.gop, chooseKind(false));
|
||||||
|
try std.testing.expectEqual(Kind.virtio, chooseKind(true));
|
||||||
|
}
|
||||||
@@ -0,0 +1,192 @@
|
|||||||
|
//! The compositor's pure core: rectangle math and the three blitting primitives the
|
||||||
|
//! display service composes frames from — fill a rectangle of a surface, composite one
|
||||||
|
//! surface onto another clipped to a damage rectangle, and copy a client-supplied pixel
|
||||||
|
//! tile in. Deliberately free of any syscall or `runtime` dependency (it takes plain
|
||||||
|
//! pixel pointers), so it is host-tested under `zig build test`. The service
|
||||||
|
//! (system/services/display/display.zig) wires real mmap'd surfaces and the framebuffer
|
||||||
|
//! to it. Pixels are opaque native 32-bit values — v1 layers don't alpha-blend, and
|
||||||
|
//! channel order (rgbx/bgrx) is the caller's concern (see protocol.pack).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// An axis-aligned rectangle in pixels. Signed, so a surface partly off-screen (a layer
|
||||||
|
/// dragged past an edge) clips with plain arithmetic. Half-open: covers [x, x+w) × [y, y+h).
|
||||||
|
pub const Rect = struct {
|
||||||
|
x: i32,
|
||||||
|
y: i32,
|
||||||
|
w: i32,
|
||||||
|
h: i32,
|
||||||
|
|
||||||
|
pub const empty = Rect{ .x = 0, .y = 0, .w = 0, .h = 0 };
|
||||||
|
|
||||||
|
pub fn init(x: i32, y: i32, w: i32, h: i32) Rect {
|
||||||
|
return .{ .x = x, .y = y, .w = w, .h = h };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn isEmpty(r: Rect) bool {
|
||||||
|
return r.w <= 0 or r.h <= 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn right(r: Rect) i32 {
|
||||||
|
return r.x + r.w;
|
||||||
|
}
|
||||||
|
pub fn bottom(r: Rect) i32 {
|
||||||
|
return r.y + r.h;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The overlap of two rectangles, or an empty rectangle if they don't touch.
|
||||||
|
pub fn intersect(a: Rect, b: Rect) Rect {
|
||||||
|
const x0 = @max(a.x, b.x);
|
||||||
|
const y0 = @max(a.y, b.y);
|
||||||
|
const x1 = @min(a.right(), b.right());
|
||||||
|
const y1 = @min(a.bottom(), b.bottom());
|
||||||
|
return .{ .x = x0, .y = y0, .w = x1 - x0, .h = y1 - y0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bounding box of two rectangles. An empty operand contributes nothing (returns
|
||||||
|
/// the other), so folding damage rectangles with `unite` from `empty` yields their
|
||||||
|
/// bounding box.
|
||||||
|
pub fn unite(a: Rect, b: Rect) Rect {
|
||||||
|
if (a.isEmpty()) return b;
|
||||||
|
if (b.isEmpty()) return a;
|
||||||
|
const x0 = @min(a.x, b.x);
|
||||||
|
const y0 = @min(a.y, b.y);
|
||||||
|
const x1 = @max(a.right(), b.right());
|
||||||
|
const y1 = @max(a.bottom(), b.bottom());
|
||||||
|
return .{ .x = x0, .y = y0, .w = x1 - x0, .h = y1 - y0 };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A block of 32-bit pixels: `pixels` addressed row-major with `stride` pixels between
|
||||||
|
/// row starts (≥ width — the framebuffer's stride is pitch/4, a layer's is its width).
|
||||||
|
pub const Surface = struct {
|
||||||
|
pixels: [*]u32,
|
||||||
|
stride: u32, // pixels per row
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
|
||||||
|
pub fn bounds(s: Surface) Rect {
|
||||||
|
return .{ .x = 0, .y = 0, .w = @intCast(s.width), .h = @intCast(s.height) };
|
||||||
|
}
|
||||||
|
|
||||||
|
inline fn row(s: Surface, y: u32) [*]u32 {
|
||||||
|
return s.pixels + @as(usize, y) * s.stride;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Fill `rect` of `s` with the native pixel `colour`, clipped to `s`'s bounds.
|
||||||
|
pub fn fillRect(s: Surface, rect: Rect, colour: u32) void {
|
||||||
|
const c = rect.intersect(s.bounds());
|
||||||
|
if (c.isEmpty()) return;
|
||||||
|
var y: i32 = c.y;
|
||||||
|
while (y < c.bottom()) : (y += 1) {
|
||||||
|
const r = s.row(@intCast(y));
|
||||||
|
var x: i32 = c.x;
|
||||||
|
while (x < c.right()) : (x += 1) r[@intCast(x)] = colour;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Composite the whole of `layer` onto `dst` with the layer's top-left at (`dx`, `dy`),
|
||||||
|
/// painting only the pixels that fall inside `clip` (a `dst`-space rectangle) and inside
|
||||||
|
/// `dst`. Opaque copy. This is the primitive `present` repeats over the visible layer
|
||||||
|
/// stack, bottom to top, for each damaged region.
|
||||||
|
pub fn composite(dst: Surface, dx: i32, dy: i32, layer: Surface, clip: Rect) void {
|
||||||
|
const on_screen = Rect{ .x = dx, .y = dy, .w = @intCast(layer.width), .h = @intCast(layer.height) };
|
||||||
|
const region = on_screen.intersect(clip).intersect(dst.bounds());
|
||||||
|
if (region.isEmpty()) return;
|
||||||
|
var y: i32 = region.y;
|
||||||
|
while (y < region.bottom()) : (y += 1) {
|
||||||
|
const src = layer.row(@intCast(y - dy));
|
||||||
|
const d = dst.row(@intCast(y));
|
||||||
|
var x: i32 = region.x;
|
||||||
|
while (x < region.right()) : (x += 1) {
|
||||||
|
d[@intCast(x)] = src[@intCast(x - dx)];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy a `w`×`h` tile of native pixels from `src` (raw little-endian bytes, row-major,
|
||||||
|
/// tightly packed) into `dst` at (`dx`, `dy`), clipped to `dst`'s bounds. `src` is read
|
||||||
|
/// with `readInt` because it comes straight out of an IPC message buffer and carries no
|
||||||
|
/// alignment guarantee. Returns without touching anything if `src` is short.
|
||||||
|
pub fn blitTile(dst: Surface, dx: i32, dy: i32, src: []const u8, w: u32, h: u32) void {
|
||||||
|
if (src.len < @as(usize, w) * h * 4) return;
|
||||||
|
var ty: u32 = 0;
|
||||||
|
while (ty < h) : (ty += 1) {
|
||||||
|
const yy = dy + @as(i32, @intCast(ty));
|
||||||
|
if (yy < 0 or yy >= dst.height) continue;
|
||||||
|
const drow = dst.row(@intCast(yy));
|
||||||
|
var tx: u32 = 0;
|
||||||
|
while (tx < w) : (tx += 1) {
|
||||||
|
const xx = dx + @as(i32, @intCast(tx));
|
||||||
|
if (xx < 0 or xx >= dst.width) continue;
|
||||||
|
const off = (@as(usize, ty) * w + tx) * 4;
|
||||||
|
drow[@intCast(xx)] = std.mem.readInt(u32, src[off..][0..4], .little);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
|
test "rect intersect: overlap and disjoint" {
|
||||||
|
try std.testing.expectEqual(Rect.init(5, 5, 5, 5), Rect.init(0, 0, 10, 10).intersect(Rect.init(5, 5, 10, 10)));
|
||||||
|
try std.testing.expect(Rect.init(0, 0, 10, 10).intersect(Rect.init(20, 20, 5, 5)).isEmpty());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "rect unite: bounding box, empty is identity" {
|
||||||
|
const a = Rect.init(2, 2, 4, 4);
|
||||||
|
try std.testing.expectEqual(Rect.init(2, 1, 10, 5), a.unite(Rect.init(10, 1, 2, 2)));
|
||||||
|
try std.testing.expectEqual(a, a.unite(Rect.empty));
|
||||||
|
try std.testing.expectEqual(a, Rect.empty.unite(a));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "fillRect clips to surface and honours stride padding" {
|
||||||
|
// A 4×3 surface inside a 6-wide allocation (stride 6 > width 4), like pitch padding.
|
||||||
|
var mem = [_]u32{0} ** (6 * 3);
|
||||||
|
const s = Surface{ .pixels = &mem, .stride = 6, .width = 4, .height = 3 };
|
||||||
|
fillRect(s, Rect.init(-1, -1, 3, 3), 0xAB); // straddles the top-left corner
|
||||||
|
try std.testing.expectEqual(@as(u32, 0xAB), mem[0 * 6 + 0]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0xAB), mem[1 * 6 + 1]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), mem[0 * 6 + 2]); // beyond the 2-wide fill
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), mem[2 * 6 + 0]); // row 2 untouched
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), mem[0 * 6 + 4]); // stride padding untouched
|
||||||
|
}
|
||||||
|
|
||||||
|
test "composite: overlap shows the top layer, clipped to damage" {
|
||||||
|
var back = [_]u32{0} ** (8 * 8);
|
||||||
|
const dst = Surface{ .pixels = &back, .stride = 8, .width = 8, .height = 8 };
|
||||||
|
var lo = [_]u32{0x11} ** (4 * 4);
|
||||||
|
var hi = [_]u32{0x22} ** (4 * 4);
|
||||||
|
const low = Surface{ .pixels = &lo, .stride = 4, .width = 4, .height = 4 };
|
||||||
|
const high = Surface{ .pixels = &hi, .stride = 4, .width = 4, .height = 4 };
|
||||||
|
composite(dst, 0, 0, low, dst.bounds()); // bottom at (0,0)
|
||||||
|
composite(dst, 2, 2, high, dst.bounds()); // top overlaps at (2,2)
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x11), back[0 * 8 + 0]); // bottom-only
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x22), back[3 * 8 + 3]); // overlap → top wins
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x22), back[5 * 8 + 5]); // top-only
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[7 * 8 + 7]); // neither
|
||||||
|
}
|
||||||
|
|
||||||
|
test "composite honours the damage rectangle" {
|
||||||
|
var back = [_]u32{0} ** (8 * 8);
|
||||||
|
const dst = Surface{ .pixels = &back, .stride = 8, .width = 8, .height = 8 };
|
||||||
|
var fill = [_]u32{0x33} ** (8 * 8);
|
||||||
|
const layer = Surface{ .pixels = &fill, .stride = 8, .width = 8, .height = 8 };
|
||||||
|
composite(dst, 0, 0, layer, Rect.init(2, 2, 2, 2)); // only this damage region
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x33), back[2 * 8 + 2]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x33), back[3 * 8 + 3]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[1 * 8 + 1]); // outside damage
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[4 * 8 + 4]); // outside damage
|
||||||
|
}
|
||||||
|
|
||||||
|
test "blitTile copies a packed tile, clipping and reading unaligned bytes" {
|
||||||
|
var back = [_]u32{0} ** (4 * 4);
|
||||||
|
const dst = Surface{ .pixels = &back, .stride = 4, .width = 4, .height = 4 };
|
||||||
|
// A 2×2 tile in a byte buffer offset by one byte, so reads are unaligned.
|
||||||
|
var raw = [_]u8{0} ** (1 + 2 * 2 * 4);
|
||||||
|
const tile = raw[1..];
|
||||||
|
for (0..4) |i| std.mem.writeInt(u32, tile[i * 4 ..][0..4], @intCast(0xA0 + i), .little);
|
||||||
|
blitTile(dst, 3, 3, tile, 2, 2); // bottom-right corner; only (3,3) lands on-surface
|
||||||
|
try std.testing.expectEqual(@as(u32, 0xA0), back[3 * 4 + 3]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[0]); // nothing else touched
|
||||||
|
}
|
||||||
@@ -0,0 +1,462 @@
|
|||||||
|
//! /system/services/display — the display service (docs/display.md, docs/display-v2.md).
|
||||||
|
//! A ring-3 compositor: it composes an ordered stack of **layers** into a cacheable
|
||||||
|
//! surface and presents finished frames. Scanout — how a frame reaches the panel — is a
|
||||||
|
//! pluggable **backend** ([backend.zig](backend.zig)): the GOP framebuffer today, a native
|
||||||
|
//! virtio-gpu driver later; this file never learns which is active. It owns the layer stack
|
||||||
|
//! and damage tracking; the pixel math is the pure, host-tested
|
||||||
|
//! [compositor.zig](compositor.zig).
|
||||||
|
//!
|
||||||
|
//! A layer is a server-owned surface (its own cacheable buffer) with a screen position,
|
||||||
|
//! z-order, and visibility. Clients create layers, draw into them by command (`fill_rect`,
|
||||||
|
//! `blit_tile`), mark `damage`, and ask for a `present`; the compositor repaints only the
|
||||||
|
//! damaged region — clear it, paint the visible layers bottom-to-top into the backend's
|
||||||
|
//! surface, then `backend.present(damage)`. Shared-memory client surfaces are later
|
||||||
|
//! (docs/display-v2.md).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const compositor = @import("compositor.zig");
|
||||||
|
const backend_mod = @import("backend.zig");
|
||||||
|
|
||||||
|
const protocol = runtime.display_protocol;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const system = runtime.system;
|
||||||
|
const Rect = compositor.Rect;
|
||||||
|
const Surface = compositor.Surface;
|
||||||
|
|
||||||
|
/// The active scanout backend — the GOP framebuffer at boot, upgraded to a native driver
|
||||||
|
/// (virtio-gpu) when one announces itself (V4).
|
||||||
|
var backend: backend_mod.Backend = undefined;
|
||||||
|
var frames: u64 = 0;
|
||||||
|
|
||||||
|
/// This service's endpoint, kept so `attach_scanout` can arm a one-shot timer: the very first
|
||||||
|
/// native present must happen in a *later* loop iteration, after the reply to the driver's
|
||||||
|
/// announce has unblocked it and it is serving its `.scanout` channel — presenting inline
|
||||||
|
/// would deadlock (we'd call the driver while it waits on our reply).
|
||||||
|
var service_endpoint: ipc.Handle = 0;
|
||||||
|
|
||||||
|
/// Set when the backend has just been upgraded to virtio-gpu: the next present repaints the
|
||||||
|
/// whole screen into the shared surface and reads a pixel back to confirm the frame landed.
|
||||||
|
var pending_native_verify: bool = false;
|
||||||
|
|
||||||
|
/// Set alongside it: after the native present is verified, run the mode-set self-check once
|
||||||
|
/// (query the driver's modes, switch to a different one, confirm the geometry changed) — the
|
||||||
|
/// serial proof the runtime-resolution-change + fenced-present paths work (V5).
|
||||||
|
var pending_modeset_check: bool = false;
|
||||||
|
|
||||||
|
/// The wallpaper the compositor clears damaged regions to before painting layers.
|
||||||
|
var background: u32 = 0;
|
||||||
|
|
||||||
|
/// The layer stack. A fixed table (a compositor has few top-level surfaces during
|
||||||
|
/// bring-up); each used slot owns an mmap'd surface. `damage` accumulates the dirty
|
||||||
|
/// screen region since the last `present`, so a present touches only what changed.
|
||||||
|
const maximum_layers = 16;
|
||||||
|
|
||||||
|
const Layer = struct {
|
||||||
|
used: bool = false,
|
||||||
|
x: i32 = 0,
|
||||||
|
y: i32 = 0,
|
||||||
|
z: u32 = 0,
|
||||||
|
visible: bool = false,
|
||||||
|
surface: Surface = undefined,
|
||||||
|
surface_len: usize = 0, // for munmap on destroy
|
||||||
|
};
|
||||||
|
|
||||||
|
var layers: [maximum_layers]Layer = [_]Layer{.{}} ** maximum_layers;
|
||||||
|
var damage: Rect = Rect.empty;
|
||||||
|
|
||||||
|
// --- geometry helpers -------------------------------------------------------
|
||||||
|
|
||||||
|
fn screenRect() Rect {
|
||||||
|
const m = backend.info();
|
||||||
|
return .{ .x = 0, .y = 0, .w = @intCast(m.width), .h = @intCast(m.height) };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn layerScreenRect(l: *const Layer) Rect {
|
||||||
|
return .{ .x = l.x, .y = l.y, .w = @intCast(l.surface.width), .h = @intCast(l.surface.height) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Add `r` (screen coordinates) to the pending damage, clipped to the screen.
|
||||||
|
fn addDamage(r: Rect) void {
|
||||||
|
damage = damage.unite(r.intersect(screenRect()));
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- layer operations (called from onMessage and the self-check) ------------
|
||||||
|
|
||||||
|
fn freeLayer() ?u32 {
|
||||||
|
for (&layers, 0..) |*l, i| {
|
||||||
|
if (!l.used) return @intCast(i);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A used layer by id, or null if the id is out of range or free.
|
||||||
|
fn layerAt(id: u32) ?*Layer {
|
||||||
|
if (id >= maximum_layers or !layers[id].used) return null;
|
||||||
|
return &layers[id];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32, visible: bool) ?u32 {
|
||||||
|
if (w == 0 or h == 0) return null;
|
||||||
|
const slot = freeLayer() orelse return null;
|
||||||
|
const len = @as(usize, w) * h * 4;
|
||||||
|
const base = system.mmap(len, system.PROT_READ | system.PROT_WRITE);
|
||||||
|
if (system.mmapFailed(base)) return null;
|
||||||
|
layers[slot] = .{
|
||||||
|
.used = true,
|
||||||
|
.x = x,
|
||||||
|
.y = y,
|
||||||
|
.z = z,
|
||||||
|
.visible = visible,
|
||||||
|
.surface = .{ .pixels = @ptrFromInt(base), .stride = w, .width = w, .height = h },
|
||||||
|
.surface_len = len,
|
||||||
|
};
|
||||||
|
return slot;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fillLayer(id: u32, local: Rect, colour: u32) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
compositor.fillRect(l.surface, local, colour);
|
||||||
|
// Damage in screen space = the fill, translated by the layer origin, within the layer.
|
||||||
|
const screen = Rect{ .x = l.x + local.x, .y = l.y + local.y, .w = local.w, .h = local.h };
|
||||||
|
addDamage(screen.intersect(layerScreenRect(l)));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn blitLayer(id: u32, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
compositor.blitTile(l.surface, x, y, pixels, w, h);
|
||||||
|
const screen = Rect{ .x = l.x + x, .y = l.y + y, .w = @intCast(w), .h = @intCast(h) };
|
||||||
|
addDamage(screen.intersect(layerScreenRect(l)));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn configureLayer(id: u32, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
addDamage(layerScreenRect(l)); // the old footprint must repaint
|
||||||
|
l.x = x;
|
||||||
|
l.y = y;
|
||||||
|
l.z = z;
|
||||||
|
l.visible = visible;
|
||||||
|
addDamage(layerScreenRect(l)); // and the new one
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn destroyLayer(id: u32) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
addDamage(layerScreenRect(l));
|
||||||
|
_ = system.munmap(@intFromPtr(l.surface.pixels), l.surface_len);
|
||||||
|
l.* = .{};
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- compositing + present --------------------------------------------------
|
||||||
|
|
||||||
|
/// Repaint the damaged region `clip` of the backend's compose surface: clear it to the
|
||||||
|
/// background, then paint every visible layer that overlaps it, bottom to top (ascending z).
|
||||||
|
fn compositeInto(clip: Rect) void {
|
||||||
|
const target = backend.surface();
|
||||||
|
compositor.fillRect(target, clip, background);
|
||||||
|
|
||||||
|
// z-order the used, visible layers (n ≤ 16; a plain insertion sort of indices).
|
||||||
|
var order: [maximum_layers]u32 = undefined;
|
||||||
|
var n: usize = 0;
|
||||||
|
for (layers, 0..) |l, i| {
|
||||||
|
if (l.used and l.visible) {
|
||||||
|
order[n] = @intCast(i);
|
||||||
|
n += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var a: usize = 1;
|
||||||
|
while (a < n) : (a += 1) {
|
||||||
|
const key = order[a];
|
||||||
|
var b: usize = a;
|
||||||
|
while (b > 0 and layers[order[b - 1]].z > layers[key].z) : (b -= 1) order[b] = order[b - 1];
|
||||||
|
order[b] = key;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (order[0..n]) |i| {
|
||||||
|
const l = layers[i];
|
||||||
|
compositor.composite(target, l.x, l.y, l.surface, clip);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Composite the accumulated damage into the backend's surface, hand it to the backend to
|
||||||
|
/// put on screen, then clear the damage. A no-op when nothing is dirty. The frame counter
|
||||||
|
/// advances regardless, so callers can name frames.
|
||||||
|
fn present() void {
|
||||||
|
const dirty = damage.intersect(screenRect());
|
||||||
|
if (!dirty.isEmpty()) {
|
||||||
|
compositeInto(dirty);
|
||||||
|
backend.present(dirty);
|
||||||
|
}
|
||||||
|
damage = Rect.empty;
|
||||||
|
frames += 1;
|
||||||
|
|
||||||
|
// The first present after a native upgrade confirms the composited frame actually reached
|
||||||
|
// the shared scanout surface (the automated stand-in for "it's on screen").
|
||||||
|
if (pending_native_verify and !dirty.isEmpty()) {
|
||||||
|
pending_native_verify = false;
|
||||||
|
verifyNativePresent();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a pixel straight back from the shared scanout surface after a native present. The
|
||||||
|
/// surface starts zeroed, so a non-zero centre pixel means the compositor wrote the frame into
|
||||||
|
/// the pages the driver transfers-and-flushes from — that, plus the driver acking the present
|
||||||
|
/// over `.scanout`, is the serial proof the native path works.
|
||||||
|
fn verifyNativePresent() void {
|
||||||
|
const s = backend.surface();
|
||||||
|
const sample = s.pixels[@as(usize, s.height / 2) * s.stride + s.width / 2];
|
||||||
|
if (sample != 0) {
|
||||||
|
_ = system.write("display: native present verified\n");
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: native present FAILED (blank surface)\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A native scanout driver announced itself: map the shared surface it handed over, find its
|
||||||
|
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
||||||
|
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
||||||
|
/// unblocks the driver and it starts serving `.scanout`.
|
||||||
|
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
||||||
|
const cap = capability orelse return fail(reply);
|
||||||
|
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
||||||
|
const mapped = runtime.shm.map(cap) orelse return fail(reply);
|
||||||
|
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
|
||||||
|
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
||||||
|
// scanout. (The previous shared mapping leaks — there is no shm_unmap syscall yet — but the
|
||||||
|
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
||||||
|
const reattach = switch (backend) {
|
||||||
|
.virtio => true,
|
||||||
|
else => false,
|
||||||
|
};
|
||||||
|
|
||||||
|
backend = .{ .virtio = .{
|
||||||
|
.pixels = @ptrCast(@alignCast(mapped)),
|
||||||
|
.stride = stride,
|
||||||
|
.width = width,
|
||||||
|
.height = height,
|
||||||
|
.format = format,
|
||||||
|
.scanout = scanout,
|
||||||
|
} };
|
||||||
|
background = protocol.pack(format, 0x20, 0x30, 0x48); // re-pack the wallpaper for the mode
|
||||||
|
addDamage(screenRect()); // the whole new surface must be painted
|
||||||
|
pending_native_verify = true;
|
||||||
|
if (!reattach) pending_modeset_check = true; // the mode-set self-check runs once, on first upgrade
|
||||||
|
_ = system.timerOnce(service_endpoint, 50); // present once the driver is serving .scanout
|
||||||
|
_ = system.write(if (reattach)
|
||||||
|
"display: scanout re-attached\n"
|
||||||
|
else
|
||||||
|
"display: scanout upgraded to virtio-gpu\n");
|
||||||
|
return ok(reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// After the native upgrade is verified, prove the runtime-resolution-change and fenced-present
|
||||||
|
/// paths: query the driver's modes, switch to one that differs from the current, re-composite
|
||||||
|
/// the whole screen at the new size, and confirm the backend now reports that geometry. The
|
||||||
|
/// present goes through the driver's fenced flush, so a clean present is a vsync present.
|
||||||
|
fn modesetSelfCheck() void {
|
||||||
|
if (!backend.canModeSet()) return;
|
||||||
|
var mode_list: [4]backend_mod.Mode = undefined;
|
||||||
|
const count = backend.modes(&mode_list);
|
||||||
|
if (count == 0) {
|
||||||
|
_ = system.write("display: mode-set self-check: no modes reported\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const current = backend.info();
|
||||||
|
var target: ?backend_mod.Mode = null;
|
||||||
|
for (mode_list[0..count]) |m| {
|
||||||
|
if (m.width != current.width or m.height != current.height) {
|
||||||
|
target = m;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const wanted = target orelse {
|
||||||
|
_ = system.write("display: mode-set self-check: no alternate mode offered\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (!backend.setMode(wanted.width, wanted.height)) {
|
||||||
|
_ = system.write("display: mode set FAILED\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
addDamage(screenRect()); // repaint the whole screen at the new resolution, then present it
|
||||||
|
present();
|
||||||
|
|
||||||
|
const now = backend.info();
|
||||||
|
if (now.width == wanted.width and now.height == wanted.height) {
|
||||||
|
var line: [80]u8 = undefined;
|
||||||
|
_ = system.write(std.fmt.bufPrint(&line, "display: mode set to {d}x{d}, verified\n", .{ now.width, now.height }) catch "display: mode set, verified\n");
|
||||||
|
if (backend.hasVsync()) _ = system.write("display: vsync present ok\n");
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: mode set FAILED (geometry unchanged)\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- startup self-check -----------------------------------------------------
|
||||||
|
|
||||||
|
/// Prove the compositor wiring on the real backend: two overlapping opaque layers,
|
||||||
|
/// composited, must show the top layer in the overlap and the bottom layer outside it.
|
||||||
|
/// Exercises the whole path — mmap surfaces, the z-sort, damage, composite into the
|
||||||
|
/// backend surface — and reads the composited result back. Cleans up after itself.
|
||||||
|
fn selfCheck() void {
|
||||||
|
const format = backend.info().format;
|
||||||
|
const red = protocol.pack(format, 0xC0, 0x20, 0x20);
|
||||||
|
const green = protocol.pack(format, 0x20, 0xC0, 0x20);
|
||||||
|
const bottom = createLayer(100, 100, 80, 80, 0, true) orelse return fail_check("create");
|
||||||
|
const top = createLayer(140, 140, 80, 80, 1, true) orelse return fail_check("create");
|
||||||
|
_ = fillLayer(bottom, Rect.init(0, 0, 80, 80), red);
|
||||||
|
_ = fillLayer(top, Rect.init(0, 0, 80, 80), green);
|
||||||
|
present();
|
||||||
|
|
||||||
|
const surface = backend.surface();
|
||||||
|
const overlap = surface.pixels[@as(usize, 150) * surface.stride + 150]; // in both → top
|
||||||
|
const bottom_only = surface.pixels[@as(usize, 110) * surface.stride + 110]; // bottom only
|
||||||
|
|
||||||
|
_ = destroyLayer(top);
|
||||||
|
_ = destroyLayer(bottom);
|
||||||
|
present(); // repaint the self-check region back to the background
|
||||||
|
|
||||||
|
if (overlap == green and bottom_only == red) {
|
||||||
|
_ = system.write("display: compositor self-check ok\n");
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: compositor self-check FAILED\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fail_check(_: []const u8) void {
|
||||||
|
_ = system.write("display: compositor self-check FAILED (setup)\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- service ----------------------------------------------------------------
|
||||||
|
|
||||||
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
|
service_endpoint = endpoint;
|
||||||
|
|
||||||
|
// Pick the scanout backend (GOP today). It logs the reason on failure.
|
||||||
|
backend = backend_mod.select() orelse return false;
|
||||||
|
const mode = backend.info();
|
||||||
|
background = protocol.pack(mode.format, 0x20, 0x30, 0x48); // a dark slate wallpaper
|
||||||
|
|
||||||
|
// Clear the whole screen through the compose surface → present path (double buffering:
|
||||||
|
// no direct-to-scanout drawing).
|
||||||
|
addDamage(screenRect());
|
||||||
|
present();
|
||||||
|
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = system.write(std.fmt.bufPrint(&line, "display: online {d}x{d} pitch {d} format {d}\n", .{
|
||||||
|
mode.width, mode.height, mode.pitch, mode.format,
|
||||||
|
}) catch "display: online\n");
|
||||||
|
_ = system.write("display: presented frame 0\n");
|
||||||
|
|
||||||
|
selfCheck();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeReply(reply: []u8, value: protocol.Reply) usize {
|
||||||
|
const bytes = std.mem.asBytes(&value);
|
||||||
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
|
return bytes.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn ok(reply: []u8) usize {
|
||||||
|
return writeReply(reply, .{ .status = 0 });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fail(reply: []u8) usize {
|
||||||
|
return writeReply(reply, .{ .status = -1 });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = sender;
|
||||||
|
if (message.len < protocol.request_size) return fail(reply);
|
||||||
|
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||||
|
const payload = message[protocol.request_size..];
|
||||||
|
// Switch on the raw operation value — an out-of-range one must fail cleanly, not
|
||||||
|
// panic an `@enumFromInt`.
|
||||||
|
switch (request.operation) {
|
||||||
|
@intFromEnum(protocol.Operation.info) => {
|
||||||
|
const m = backend.info();
|
||||||
|
return writeReply(reply, .{ .status = 0, .width = m.width, .height = m.height, .pitch = m.pitch, .format = m.format });
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.create_layer) => {
|
||||||
|
// x/y are signed coordinates carried in the u32 wire fields — reinterpret the
|
||||||
|
// bits (@bitCast), don't range-check (@intCast) which a negative would fail.
|
||||||
|
const slot = createLayer(@bitCast(request.x), @bitCast(request.y), request.width, request.height, request.z, request.visible != 0) orelse return fail(reply);
|
||||||
|
return writeReply(reply, .{ .status = 0, .layer = slot });
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.configure_layer) => {
|
||||||
|
return if (configureLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.z, request.visible != 0)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.destroy_layer) => {
|
||||||
|
return if (destroyLayer(request.layer)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.fill_rect) => {
|
||||||
|
const local = Rect.init(@bitCast(request.x), @bitCast(request.y), @intCast(request.width), @intCast(request.height));
|
||||||
|
return if (fillLayer(request.layer, local, request.colour)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.blit_tile) => {
|
||||||
|
return if (blitLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.width, request.height, payload)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.damage) => {
|
||||||
|
const l = layerAt(request.layer) orelse return fail(reply);
|
||||||
|
const screen = Rect{ .x = l.x + @as(i32, @bitCast(request.x)), .y = l.y + @as(i32, @bitCast(request.y)), .w = @intCast(request.width), .h = @intCast(request.height) };
|
||||||
|
addDamage(screen.intersect(layerScreenRect(l)));
|
||||||
|
return ok(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.present) => {
|
||||||
|
present();
|
||||||
|
return ok(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.attach_scanout) => {
|
||||||
|
return attachScanout(request.x, request.width, request.height, request.colour, capability, reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.set_mode) => {
|
||||||
|
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
||||||
|
addDamage(screenRect()); // repaint the whole screen at the new resolution
|
||||||
|
present();
|
||||||
|
return ok(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.get_modes) => {
|
||||||
|
var list: [4]backend_mod.Mode = undefined;
|
||||||
|
const count = backend.modes(&list);
|
||||||
|
var response = protocol.ModesReply{ .status = 0, .count = @intCast(count), .modes = undefined };
|
||||||
|
for (0..protocol.max_modes) |i| {
|
||||||
|
response.modes[i] = if (i < count)
|
||||||
|
.{ .width = list[i].width, .height = list[i].height }
|
||||||
|
else
|
||||||
|
.{ .width = 0, .height = 0 };
|
||||||
|
}
|
||||||
|
const bytes = std.mem.asBytes(&response);
|
||||||
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
|
return bytes.len;
|
||||||
|
},
|
||||||
|
else => return fail(reply),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The only notification the compositor arms is the post-attach present timer: repaint the
|
||||||
|
/// screen into the freshly attached native surface, verify the frame landed, then run the
|
||||||
|
/// one-shot mode-set self-check (V5).
|
||||||
|
fn onNotification(badge: u64) void {
|
||||||
|
_ = badge;
|
||||||
|
present(); // native present + verify (first timer fire after the upgrade)
|
||||||
|
if (pending_modeset_check) {
|
||||||
|
pending_modeset_check = false;
|
||||||
|
modesetSelfCheck();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
.service = .display,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
.on_notification = onNotification,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,113 @@
|
|||||||
|
//! The display wire protocol — what a client says to the display service over its
|
||||||
|
//! well-known `.display` endpoint. extern-struct messages with an `Operation` tag, the
|
||||||
|
//! same shape as block/vfs/input protocols. The compositor owns the framebuffer and an
|
||||||
|
//! ordered stack of **layers**; a client creates layers, draws into them with these
|
||||||
|
//! operations, marks damage, and asks for a `present`. v1 surfaces are server-owned (a
|
||||||
|
//! client draws by command); shared-memory surfaces are a later milestone (docs/display.md).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub const Operation = enum(u32) {
|
||||||
|
/// info() -> { width, height, pitch, format }: the display's current mode.
|
||||||
|
info = 0,
|
||||||
|
/// create_layer(x, y, width, height, z) -> { layer }: a new server-owned surface.
|
||||||
|
create_layer = 1,
|
||||||
|
/// configure_layer(layer, x, y, z, visible): move, restack, show, or hide a layer.
|
||||||
|
configure_layer = 2,
|
||||||
|
/// destroy_layer(layer): release a layer.
|
||||||
|
destroy_layer = 3,
|
||||||
|
/// fill_rect(layer, x, y, width, height, colour): fill a rectangle of a layer.
|
||||||
|
fill_rect = 4,
|
||||||
|
/// blit_tile(layer, x, y, width, height, <inline pixels>): copy a small pixel tile in.
|
||||||
|
blit_tile = 5,
|
||||||
|
/// damage(layer, x, y, width, height): mark a region dirty for the next present.
|
||||||
|
damage = 6,
|
||||||
|
/// present(): composite the dirty layers and flush to the screen.
|
||||||
|
present = 7,
|
||||||
|
/// attach_scanout(x=stride, width, height, colour=format) + <surface capability>: a native
|
||||||
|
/// scanout driver announces itself, handing over the shared scanout surface as an `ipc_call`
|
||||||
|
/// send_cap. The compositor maps it, looks up the driver's `.scanout` present channel, and
|
||||||
|
/// upgrades off the GOP floor (docs/display-v2.md V4). `x` is the surface's row stride in
|
||||||
|
/// pixels, `colour` the DisplayFormat.
|
||||||
|
attach_scanout = 8,
|
||||||
|
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||||
|
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||||
|
set_mode = 9,
|
||||||
|
/// get_modes() -> ModesReply: the resolutions the display can switch to (empty on GOP).
|
||||||
|
get_modes = 10,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The fixed request header. A `blit_tile`'s pixel payload (width*height 32-bit pixels)
|
||||||
|
/// follows this header inline in the same message, up to `maximum_payload`.
|
||||||
|
pub const Request = extern struct {
|
||||||
|
operation: u32,
|
||||||
|
layer: u32 = 0, // create/configure/destroy/fill/blit/damage: the target layer
|
||||||
|
x: u32 = 0,
|
||||||
|
y: u32 = 0,
|
||||||
|
width: u32 = 0,
|
||||||
|
height: u32 = 0,
|
||||||
|
z: u32 = 0, // create_layer / configure_layer: stacking order (higher = in front)
|
||||||
|
colour: u32 = 0, // fill_rect: the fill colour (native pixel value)
|
||||||
|
visible: u32 = 1, // configure_layer: 0 hides the layer
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Reply = extern struct {
|
||||||
|
status: i32, // 0 on success, negative on failure
|
||||||
|
reserved: u32 = 0,
|
||||||
|
// info():
|
||||||
|
width: u32 = 0,
|
||||||
|
height: u32 = 0,
|
||||||
|
pitch: u32 = 0,
|
||||||
|
format: u32 = 0, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||||
|
// create_layer():
|
||||||
|
layer: u32 = 0,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One selectable display mode.
|
||||||
|
pub const Mode = extern struct { width: u32, height: u32 };
|
||||||
|
pub const max_modes = 4;
|
||||||
|
|
||||||
|
/// The reply to `get_modes`: a small fixed list of resolutions the display can switch to.
|
||||||
|
pub const ModesReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
count: u32,
|
||||||
|
modes: [max_modes]Mode,
|
||||||
|
};
|
||||||
|
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||||
|
|
||||||
|
/// The IPC message size — the kernel caps every message at `MESSAGE_MAXIMUM` (256 bytes,
|
||||||
|
/// system/kernel/ipc-synchronous.zig), so this matches it (a larger receive/reply buffer
|
||||||
|
/// is rejected with -E2BIG). A `blit_tile` therefore carries only a *small* tile inline —
|
||||||
|
/// `maximum_payload` bytes = up to 54 pixels, enough for a cursor or small sprite; larger
|
||||||
|
/// bitmaps are the deferred shared-memory surface path (docs/display.md).
|
||||||
|
pub const message_maximum: usize = 256;
|
||||||
|
pub const request_size: usize = @sizeOf(Request);
|
||||||
|
pub const reply_size: usize = @sizeOf(Reply);
|
||||||
|
pub const maximum_payload: usize = message_maximum - request_size;
|
||||||
|
|
||||||
|
/// Pack an 8-bit-per-channel colour into the display's native 32-bit pixel for `format`
|
||||||
|
/// (a device-abi `DisplayFormat`: 0 = rgbx, 1 = bgrx). Shared so a `colour` in a
|
||||||
|
/// `fill_rect` request means the same thing to the client that sends it and the
|
||||||
|
/// compositor that paints it. Little-endian memory, reserved byte 0: rgbx puts red in
|
||||||
|
/// the low byte, bgrx puts blue there.
|
||||||
|
pub fn pack(format: u32, r: u8, g: u8, b: u8) u32 {
|
||||||
|
const rr: u32 = r;
|
||||||
|
const gg: u32 = g;
|
||||||
|
const bb: u32 = b;
|
||||||
|
return switch (format) {
|
||||||
|
1 => bb | (gg << 8) | (rr << 16), // bgrx
|
||||||
|
else => rr | (gg << 8) | (bb << 16), // rgbx
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "pack encodes native byte order for rgbx and bgrx" {
|
||||||
|
// rgbx: red in the low byte, blue in byte 2.
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0000_00AA), pack(0, 0xAA, 0, 0));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(0, 0, 0, 0xAA));
|
||||||
|
// bgrx: blue in the low byte, red in byte 2.
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0000_00AA), pack(1, 0, 0, 0xAA));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(1, 0xAA, 0, 0));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0000_3020), pack(0, 0x20, 0x30, 0)); // green in byte 1
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
//! The scanout wire protocol — what the compositor says to a native scanout driver (e.g.
|
||||||
|
//! virtio-gpu) over its well-known `.scanout` endpoint to put a composited frame on screen.
|
||||||
|
//! The driver owns the panel and the shared scanout surface it handed the compositor (via the
|
||||||
|
//! display service's `attach_scanout`); the compositor composites into that surface, then asks
|
||||||
|
//! the driver to present a damaged rectangle. Tiny by design — one present request. Separate
|
||||||
|
//! from the display protocol because the directions differ: clients call the compositor over
|
||||||
|
//! `.display`; the compositor calls the driver over `.scanout`. See docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub const Operation = enum(u32) {
|
||||||
|
/// present(x, y, width, height): put the given rectangle of the shared scanout surface on
|
||||||
|
/// the panel (on virtio-gpu: transfer-to-host of the region, then a fenced resource flush).
|
||||||
|
present = 0,
|
||||||
|
/// get_modes() -> ModesReply: the display modes this scanout can switch to (V5).
|
||||||
|
get_modes = 1,
|
||||||
|
/// set_mode(width, height): change the scanout resolution — the shared surface is sized to
|
||||||
|
/// the largest mode, so this just re-points the scanout rectangle; the surface is unchanged.
|
||||||
|
set_mode = 2,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Request = extern struct {
|
||||||
|
operation: u32,
|
||||||
|
x: u32 = 0,
|
||||||
|
y: u32 = 0,
|
||||||
|
width: u32 = 0,
|
||||||
|
height: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Reply = extern struct {
|
||||||
|
status: i32, // 0 on success, negative on failure
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One offered display mode.
|
||||||
|
pub const Mode = extern struct { width: u32, height: u32 };
|
||||||
|
pub const max_modes = 4;
|
||||||
|
|
||||||
|
/// The reply to `get_modes`: a small fixed list of modes.
|
||||||
|
pub const ModesReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
count: u32,
|
||||||
|
modes: [max_modes]Mode,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const message_maximum: usize = 64;
|
||||||
|
pub const request_size: usize = @sizeOf(Request);
|
||||||
|
pub const reply_size: usize = @sizeOf(Reply);
|
||||||
|
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||||
@@ -40,11 +40,16 @@ const IpcBlock = struct {
|
|||||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||||
@memcpy(destination[0..512], buffer[0..512]);
|
@memcpy(destination[0..512], buffer[0..512]);
|
||||||
return self.device.write(lba, 1, self.bounce.physical);
|
if (!self.device.write(lba, 1, self.bounce.physical)) return false;
|
||||||
|
device_dirty = true; // a block reached the device; a close will flush it
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
var ipc_block: IpcBlock = undefined;
|
var ipc_block: IpcBlock = undefined;
|
||||||
|
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||||
|
// file close — so writes are committed to stable media before a power-off.
|
||||||
|
var device_dirty: bool = false;
|
||||||
var filesystem: engine.FileSystem = undefined;
|
var filesystem: engine.FileSystem = undefined;
|
||||||
|
|
||||||
// Open handles the VFS holds against this backend: each maps a node id to a
|
// Open handles the VFS holds against this backend: each maps a node id to a
|
||||||
@@ -192,6 +197,14 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
|||||||
},
|
},
|
||||||
.close => {
|
.close => {
|
||||||
if (openAt(request.node)) |o| o.used = false;
|
if (openAt(request.node)) |o| o.used = false;
|
||||||
|
// Durable-on-close: if any block reached the device since the last
|
||||||
|
// flush, commit its cache to stable media now (best-effort). This is
|
||||||
|
// what makes init's shutdown log flush survive a real power-off, and is
|
||||||
|
// the right default for removable media the user may unplug.
|
||||||
|
if (device_dirty) {
|
||||||
|
_ = ipc_block.device.flush();
|
||||||
|
device_dirty = false;
|
||||||
|
}
|
||||||
return writeReply(out, .{ .status = 0 }, &.{});
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
},
|
},
|
||||||
.mkdir => {
|
.mkdir => {
|
||||||
|
|||||||
@@ -30,12 +30,22 @@ const log_path = "/mnt/usb/DANOS.LOG";
|
|||||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||||
/// manifest under /system/services instead of a hardcoded list.)
|
/// manifest under /system/services instead of a hardcoded list.)
|
||||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat" };
|
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display", "display-demo" };
|
||||||
|
|
||||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
/// The live process id of each boot service (0 = not running), indexed by its position
|
||||||
var child_count: usize = 0;
|
/// in `boot_services`, plus how many times init has restarted it. init supervises these:
|
||||||
|
/// it spawns them against `supervision_endpoint` and, on a child's death, restarts it (up
|
||||||
|
/// to `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md), the
|
||||||
|
/// service-level counterpart to the device manager's driver restarts.
|
||||||
|
var child_ids: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||||
|
var restart_counts: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||||
|
var shutting_down = false;
|
||||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||||
|
|
||||||
|
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||||
|
/// that faults immediately on every spawn doesn't respawn forever.
|
||||||
|
const maximum_restarts = 3;
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||||
// mmaps pages from the kernel and carves them with the free list), write into
|
// mmaps pages from the kernel and carves them with the free list), write into
|
||||||
@@ -63,11 +73,8 @@ pub fn main() void {
|
|||||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||||
// Best-effort and silent: each service announces its own readiness, and in
|
// Best-effort and silent: each service announces its own readiness, and in
|
||||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||||
for (boot_services) |service| {
|
for (boot_services, 0..) |service, i| {
|
||||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
|
||||||
children[child_count] = id;
|
|
||||||
child_count += 1;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
||||||
@@ -105,11 +112,47 @@ pub fn main() void {
|
|||||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// Child-exit notifications and anything else: keep waiting.
|
if (got.isChildExit()) {
|
||||||
|
restartChild(got.childProcessId());
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// Anything else: keep waiting.
|
||||||
if (got.isNotification()) continue;
|
if (got.isNotification()) continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A supervised boot service died. Find which one and restart it — unless it exited
|
||||||
|
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
|
||||||
|
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
||||||
|
/// iron rule 1); init only decides whether to bring it back.
|
||||||
|
fn restartChild(id: u32) void {
|
||||||
|
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||||
|
for (boot_services, 0..) |service, i| {
|
||||||
|
if (child_ids[i] != id) continue;
|
||||||
|
child_ids[i] = 0;
|
||||||
|
// An unknown reason (the record aged out) is treated as a crash worth restarting.
|
||||||
|
const reason = runtime.process.exitReason(id) orelse .fault;
|
||||||
|
if (reason == .exited) {
|
||||||
|
logLine("/system/services/init: {s} exited cleanly; not restarting\n", .{service});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
restart_counts[i] += 1;
|
||||||
|
if (restart_counts[i] > maximum_restarts) {
|
||||||
|
logLine("/system/services/init: {s} keeps crashing; giving up after {d} restarts\n", .{ service, maximum_restarts });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
logLine("/system/services/init: {s} died ({s}); restarting ({d}/{d})\n", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||||
|
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||||
|
}
|
||||||
|
|
||||||
|
fn logLine(comptime fmt: []const u8, args: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, args) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||||
/// call's capability) so events arrive as buffered messages here.
|
/// call's capability) so events arrive as buffered messages here.
|
||||||
fn subscribePower() void {
|
fn subscribePower() void {
|
||||||
@@ -153,15 +196,16 @@ fn flushKernelLog() void {
|
|||||||
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
||||||
/// power service to enter S5.
|
/// power service to enter S5.
|
||||||
fn shutDown() void {
|
fn shutDown() void {
|
||||||
|
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
|
||||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||||
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
||||||
// reverse-order stop loop below kills the fat server (children[3]) first, so
|
// reverse-order stop loop below kills the fat server first, so /mnt/usb must be
|
||||||
// /mnt/usb must be written while it is still mounted.
|
// written while it is still mounted.
|
||||||
flushKernelLog();
|
flushKernelLog();
|
||||||
var i = child_count;
|
var i = boot_services.len;
|
||||||
while (i > 0) {
|
while (i > 0) {
|
||||||
i -= 1;
|
i -= 1;
|
||||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
if (child_ids[i] != 0) runtime.process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||||
}
|
}
|
||||||
if (runtime.ipc.lookup(.power)) |h| {
|
if (runtime.ipc.lookup(.power)) |h| {
|
||||||
const request = power.Shutdown{};
|
const request = power.Shutdown{};
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
//! system/services/shm-client — the creating half of the shm test (docs/display-v2.md V2).
|
||||||
|
//! It `shm_create`s a shared region, writes a known pattern into it, and hands the region's
|
||||||
|
//! capability to `shm-server` as an `ipc_call` send_cap. The server maps that capability and
|
||||||
|
//! confirms the pattern is visible — proving cross-process shared memory over the extended
|
||||||
|
//! capability-passing path.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const system = runtime.system;
|
||||||
|
const shm = runtime.shm;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
|
||||||
|
const pattern_len = 4096;
|
||||||
|
|
||||||
|
/// The pattern the server checks — must match shm-server.zig.
|
||||||
|
fn expected(i: usize) u8 {
|
||||||
|
return @truncate(i *% 7 +% 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lookupServer() ?ipc.Handle {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.shm_test)) |h| return h;
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
const region = shm.create(pattern_len) orelse {
|
||||||
|
_ = system.write("shm: create failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < pattern_len) : (i += 1) region.ptr[i] = expected(i);
|
||||||
|
|
||||||
|
const server = lookupServer() orelse {
|
||||||
|
_ = system.write("shm: no server\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
// A non-empty message (so it reaches on_message, not the ping path), carrying the shm
|
||||||
|
// region's capability. The reply is empty; we just need the round trip.
|
||||||
|
var reply: [64]u8 = undefined;
|
||||||
|
_ = ipc.callCap(server, "shm", &reply, region.handle) catch {
|
||||||
|
_ = system.write("shm: call failed\n");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
//! system/services/shm-server — the receiving half of the shm test (docs/display-v2.md V2).
|
||||||
|
//! It registers under `ServiceId.shm_test`; when `shm-client` calls it carrying a
|
||||||
|
//! shared-memory capability, it `shm_map`s that capability and checks the client's pattern
|
||||||
|
//! is visible through the mapping — proving the two processes share the same physical pages
|
||||||
|
//! (not a copy). On success it prints `shm: shared 4096 bytes ok`, the test's marker.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const system = runtime.system;
|
||||||
|
const shm = runtime.shm;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
|
||||||
|
const pattern_len = 4096;
|
||||||
|
|
||||||
|
/// The pattern the client writes — must match shm-client.zig.
|
||||||
|
fn expected(i: usize) u8 {
|
||||||
|
return @truncate(i *% 7 +% 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = message;
|
||||||
|
_ = reply;
|
||||||
|
_ = sender;
|
||||||
|
const cap = capability orelse {
|
||||||
|
_ = system.write("shm: shared FAILED (no capability)\n");
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
const ptr = shm.map(cap) orelse {
|
||||||
|
_ = system.write("shm: shared FAILED (map)\n");
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < pattern_len) : (i += 1) {
|
||||||
|
if (ptr[i] != expected(i)) {
|
||||||
|
_ = system.write("shm: shared FAILED (mismatch)\n");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = system.write("shm: shared 4096 bytes ok\n");
|
||||||
|
return 0; // empty reply — the client only needs the round trip to unblock
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
runtime.service.run(64, .{ .service = .shm_test, .on_message = onMessage });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
+88
-8
@@ -165,6 +165,71 @@ CASES = [
|
|||||||
{"name": "ioport",
|
{"name": "ioport",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Display handoff (D1): the kernel seeds the loader's framebuffer as a claimable
|
||||||
|
# `display` device with a write-combining memory resource; the claim + mmio_map path
|
||||||
|
# maps it, and the leaf is genuinely write-combining (PAT entry 4), not the UC default.
|
||||||
|
{"name": "display",
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Display service (D2/D3): the user-space compositor claims the framebuffer, allocates
|
||||||
|
# a cacheable back buffer, clears it, and presents that composed frame (double-buffer
|
||||||
|
# path); then a startup self-check composites two overlapping layers and confirms the
|
||||||
|
# overlap shows the top layer (D3). Matched on the service's own heartbeats.
|
||||||
|
{"name": "display-service",
|
||||||
|
"expect": r"display: online \d+x\d+ pitch \d+[\s\S]*display: presented frame 0[\s\S]*display: compositor self-check ok",
|
||||||
|
"fail": r"display: could not|self-check FAILED|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# Display demo (D4): a separate process (display-demo) drives the compositor over the
|
||||||
|
# layer client API — wallpaper + a moving rectangle + a cursor, presented in a loop.
|
||||||
|
# `display-demo: ok` is printed only after it drove a run of frames of motion through
|
||||||
|
# the service (the visible motion is a screenshot via `zig build run-x86-64`).
|
||||||
|
{"name": "display-demo",
|
||||||
|
"expect": r"display-demo: scene up[\s\S]*display-demo: ok",
|
||||||
|
"fail": r"display-demo: (no display|create failed)|display: could not|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# Shared memory (v2 V2): shm-client creates a region, writes a pattern, and passes its
|
||||||
|
# capability to shm-server, which maps it and confirms the same bytes — proving
|
||||||
|
# cross-process shared pages over the extended capability passing.
|
||||||
|
{"name": "shm",
|
||||||
|
"expect": r"shm: shared 4096 bytes ok",
|
||||||
|
"fail": r"shm: (shared FAILED|create failed|no server|call failed|map)|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# virtio-gpu driver (v2 V3): boot with an emulated virtio-gpu. The device-manager stack
|
||||||
|
# discovers the PCI function and spawns the driver, which brings up the control virtqueue,
|
||||||
|
# creates a 2D scanout resource backed by DMA memory, set_scanouts it, paints a test
|
||||||
|
# pattern, transfers + flushes it, and waits for the device's used-ring ack, then reads
|
||||||
|
# the backing back. `scanout WxH online` + `flush acked, pixel check ok` are the markers.
|
||||||
|
{"name": "virtio-gpu",
|
||||||
|
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||||
|
"expect": r"virtio-gpu: scanout \d+x\d+ online[\s\S]*virtio-gpu: flush acked, pixel check ok",
|
||||||
|
"fail": r"virtio-gpu:.*(failed|not acked|mismatch|unable to claim|not a virtio-gpu|too small|no PCI capability|does not offer|rejected|missing common-config|not a mapped resource|could not spawn)|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# Native backend + hot-attach (v2 V4): boot the compositor + display-demo with an emulated
|
||||||
|
# virtio-gpu. The driver announces its shared scanout surface to the compositor, which maps
|
||||||
|
# it, upgrades off the GOP floor, and drives frames through the native backend — reading a
|
||||||
|
# pixel back to confirm the composited frame reached the shared surface, while the demo runs.
|
||||||
|
{"name": "display-native",
|
||||||
|
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||||
|
"mem": "512M", # boots the compositor + demo + the whole device-manager driver stack at once
|
||||||
|
# Order-independent: the demo's `ok` may print before or after the driver announces, so
|
||||||
|
# require all three markers to appear somewhere rather than in a fixed order.
|
||||||
|
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: native present verified)(?=.*display-demo: ok)",
|
||||||
|
"fail": r"display: native present FAILED|display: could not|display-demo: (no display|create failed)|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# Mode-set + EDID + vsync (v2 V5): same boot as display-native. After upgrading, the
|
||||||
|
# compositor queries the driver's modes, switches to a different resolution, and confirms the
|
||||||
|
# backend now reports it; the fenced present path makes it a vsync present. (The driver also
|
||||||
|
# logs the EDID preferred mode during bring-up.) Reuses the display-native kernel scenario.
|
||||||
|
{"name": "display-modeset",
|
||||||
|
"build_case": "display-native",
|
||||||
|
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||||
|
"mem": "512M",
|
||||||
|
"expect": r"(?s)(?=.*display: mode set to \d+x\d+, verified)(?=.*display: vsync present ok)",
|
||||||
|
"fail": r"display: mode set FAILED|display: mode-set self-check: |display: native present FAILED|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# Resilience: driver restart + re-attach (v2 V6). device-manager (in test-scanout-restart
|
||||||
|
# mode) kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||||
|
# re-announces, and the compositor re-attaches — surviving the loss. Expect the initial
|
||||||
|
# upgrade AND the re-attach; any CPU exception / panic (the compositor crashing) is a fail.
|
||||||
|
{"name": "display-reattach",
|
||||||
|
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||||
|
"mem": "512M",
|
||||||
|
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: scanout re-attached)",
|
||||||
|
"fail": r"CPU EXCEPTION|KERNEL PANIC|display: could not"},
|
||||||
# Monotonic clock (clock() syscall source): calibrated, advancing, never backwards.
|
# Monotonic clock (clock() syscall source): calibrated, advancing, never backwards.
|
||||||
{"name": "clock",
|
{"name": "clock",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
@@ -365,7 +430,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
"expect": r"acpi-parse: ok",
|
"expect": r"acpi-parse: ok",
|
||||||
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
"fail": r"acpi-parse: too few|DANOS-TEST-RESULT: FAIL"},
|
||||||
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||||
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||||
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
||||||
@@ -421,12 +486,23 @@ CASES = [
|
|||||||
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||||
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# M19.1: the ring-3 PCI scan (pci-bus walks the ECAM through its mmio_map
|
# M19.1/M19.3: the ring-3 PCI scan. pci-bus walks the ECAM through its mmio_map
|
||||||
# grant) finds exactly the functions the kernel's own walk recorded.
|
# grant and registers every function it finds; the kernel's own walk retired, so
|
||||||
|
# the broker starts empty and the driver populates it. The manager then runs the
|
||||||
|
# restart drill: ~1 s after the scan it kills pci-bus, prunes its child tree, and
|
||||||
|
# respawns it to re-claim, re-scan, and re-register the same functions. The kernel
|
||||||
|
# test asserts the broker equivalence (empty before, populated after, no
|
||||||
|
# duplicates); this ordered regex asserts the drill itself over the whole serial
|
||||||
|
# log — the backreference requires the respawn to re-scan the same count, and the
|
||||||
|
# full-capture match is immune to the transient-line races an in-kernel poll hits.
|
||||||
{"name": "pci-scan",
|
{"name": "pci-scan",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"pci-bus: (\d+) functions found[\s\S]*"
|
||||||
|
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||||
|
r"device-manager: restarting pci-bus[\s\S]*"
|
||||||
|
r"pci-bus: \1 functions found[\s\S]*"
|
||||||
|
r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
||||||
# subscribes (endpoint as capability), and observes the removed/added events
|
# subscribes (endpoint as capability), and observes the removed/added events
|
||||||
@@ -492,9 +568,8 @@ CASES = [
|
|||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
||||||
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
||||||
{"name": "poweroff",
|
# Soft-off (S5) is owned by the ring-3 acpi service now (see orderly-shutdown);
|
||||||
"expect": r"DANOS-POWER: attempting poweroff",
|
# the kernel keeps only reboot (FADT reset register, no AML).
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
|
||||||
{"name": "reboot",
|
{"name": "reboot",
|
||||||
"expect": r"DANOS-POWER: attempting reboot",
|
"expect": r"DANOS-POWER: attempting reboot",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
@@ -504,7 +579,10 @@ TIMEOUT = 30 # seconds per case
|
|||||||
|
|
||||||
|
|
||||||
def build(arch, case):
|
def build(arch, case):
|
||||||
cmd = ["zig", "build", f"-Dtest-case={case}"] + arch["zig_flags"]
|
# -Dserial: the harness asserts on markers the kernel writes to serial0, so the
|
||||||
|
# serial log sink must be compiled in. It is off by default (a flashed real-
|
||||||
|
# hardware image keeps its log in RAM instead; see build.zig / serial.zig).
|
||||||
|
cmd = ["zig", "build", f"-Dtest-case={case}", "-Dserial=true"] + arch["zig_flags"]
|
||||||
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
|
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
|
||||||
if r.returncode != 0:
|
if r.returncode != 0:
|
||||||
return r.stderr.strip() or r.stdout.strip()
|
return r.stderr.strip() or r.stdout.strip()
|
||||||
@@ -572,6 +650,8 @@ def run_case(arch, case):
|
|||||||
cmd = [arch["qemu"]] + arch["qemu_args"](arch, boot_volume, vars_fd, serial)
|
cmd = [arch["qemu"]] + arch["qemu_args"](arch, boot_volume, vars_fd, serial)
|
||||||
if case.get("smp"): # some cases need more than one core (e.g. parallelism)
|
if case.get("smp"): # some cases need more than one core (e.g. parallelism)
|
||||||
cmd += ["-smp", str(case["smp"])]
|
cmd += ["-smp", str(case["smp"])]
|
||||||
|
if case.get("mem"): # a case that boots the whole system at once needs more than the 128M floor
|
||||||
|
cmd[cmd.index("-m") + 1] = case["mem"]
|
||||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||||
cmd += case["qemu_extra"]
|
cmd += case["qemu_extra"]
|
||||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||||
|
|||||||
Reference in New Issue
Block a user