Compare commits
64
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9989ebbec7 | ||
|
|
626e3c5e9b | ||
|
|
2c63c76288 | ||
|
|
05fc1764de | ||
|
|
65bb04d890 | ||
|
|
24c49f56e1 | ||
|
|
0217662808 | ||
|
|
4231301896 | ||
|
|
58927ed7e5 | ||
|
|
6e0e0a62c6 | ||
|
|
88ad432758 | ||
|
|
9333d0572f | ||
|
|
f3342118f5 | ||
|
|
2723b6f778 | ||
|
|
a01a4f3b3d | ||
|
|
e1605e3235 | ||
|
|
1e80c57484 | ||
|
|
c3e9c59086 | ||
|
|
88ed3c5417 | ||
|
|
3fc2d5b083 | ||
|
|
105203b447 | ||
|
|
f9cf0007c5 | ||
|
|
69b018cc32 | ||
|
|
cd812cc00e | ||
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f | ||
|
|
67702fa250 | ||
|
|
8a38540312 | ||
|
|
54635eecf5 | ||
|
|
184d90c2c6 | ||
|
|
a32eed877d | ||
|
|
347a041d85 | ||
|
|
53e42837e0 | ||
|
|
f52c591f5e | ||
|
|
77d2e22ed1 | ||
|
|
a64a01a6a9 | ||
|
|
35e8921de8 | ||
|
|
3fb9d5936a | ||
|
|
452080e997 | ||
|
|
5b63a841ba | ||
|
|
4b9507bd59 | ||
|
|
7dec1b0767 | ||
|
|
45b8fd8614 | ||
|
|
dd22bfbc48 | ||
|
|
6e60daed6a | ||
|
|
dd044fb115 | ||
|
|
3ec14509a0 | ||
|
|
d71a5f25d3 | ||
|
|
2a0f17ae86 | ||
|
|
9ef61a0844 | ||
|
|
688b9101e8 | ||
|
|
1d7ba814dc | ||
|
|
8aba86b4ce | ||
|
|
a0c83f4b3f | ||
|
|
1ea48ed5d6 | ||
|
|
77a3ccd33d | ||
|
|
07da27dc39 | ||
|
|
d89657d0a4 | ||
|
|
849b4b62d4 | ||
|
|
8589bf713b |
@@ -82,6 +82,10 @@ straight into CI.
|
|||||||
Design notes explaining *why* behind the code live in
|
Design notes explaining *why* behind the code live in
|
||||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||||
|
|
||||||
|
For the hardware needed to run DanOS — minimum specs plus a plain-language guide
|
||||||
|
matching Intel/AMD CPU generations by name — see
|
||||||
|
[`docs/system-requirements.md`](docs/system-requirements.md).
|
||||||
|
|
||||||
## Logo
|
## Logo
|
||||||
|
|
||||||
San Serif Text "Dan OS" with a black karate belt around it.
|
San Serif Text "Dan OS" with a black karate belt around it.
|
||||||
|
|||||||
+15
-6
@@ -2,6 +2,7 @@ const std = @import("std");
|
|||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const build_options = @import("build_options");
|
||||||
const BootInformation = boot_handoff.BootInformation;
|
const BootInformation = boot_handoff.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
@@ -29,7 +30,7 @@ pub fn main() uefi.Status {
|
|||||||
// report the reason (boot services are still up) and park the machine so the
|
// report the reason (boot services are still up) and park the machine so the
|
||||||
// message stays on screen.
|
// message stays on screen.
|
||||||
boot() catch |err| {
|
boot() catch |err| {
|
||||||
log("\r\ndanos: boot failed: ");
|
log("\r\nEFI: boot failed: ");
|
||||||
logBytes(@errorName(err));
|
logBytes(@errorName(err));
|
||||||
log("\r\n");
|
log("\r\n");
|
||||||
while (true) asm volatile ("hlt");
|
while (true) asm volatile ("hlt");
|
||||||
@@ -65,14 +66,14 @@ fn boot() !noreturn {
|
|||||||
|
|
||||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||||
loadInit(bs, &boot_information) catch |err| {
|
loadInit(bs, &boot_information) catch |err| {
|
||||||
log("danos: no /system/services/init (");
|
log("EFI: no /system/services/init (");
|
||||||
logBytes(@errorName(err));
|
logBytes(@errorName(err));
|
||||||
log(") - booting without user space\r\n");
|
log(") - booting without user space\r\n");
|
||||||
};
|
};
|
||||||
|
|
||||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||||
log("danos: no initial_ramdisk (");
|
log("EFI: no initial_ramdisk (");
|
||||||
logBytes(@errorName(err));
|
logBytes(@errorName(err));
|
||||||
log(")\r\n");
|
log(")\r\n");
|
||||||
};
|
};
|
||||||
@@ -84,7 +85,7 @@ fn boot() !noreturn {
|
|||||||
// the map and exiting would invalidate the map key.
|
// the map and exiting would invalidate the map key.
|
||||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
log("danos: kernel loaded, exiting boot services\r\n");
|
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||||
boot_information.memory_map = try exitBootServices(bs);
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
@@ -395,7 +396,7 @@ fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !
|
|||||||
const image = try loadFile(bs, init_file_name);
|
const image = try loadFile(bs, init_file_name);
|
||||||
boot_information.init_base = @intFromPtr(image.ptr);
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
boot_information.init_len = image.len;
|
boot_information.init_len = image.len;
|
||||||
log("danos: /system/services/init loaded\r\n");
|
progress("EFI: /system/services/init loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||||
@@ -403,7 +404,7 @@ fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInfo
|
|||||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||||
boot_information.initial_ramdisk_len = image.len;
|
boot_information.initial_ramdisk_len = image.len;
|
||||||
log("danos: initial_ramdisk loaded\r\n");
|
progress("EFI: initial_ramdisk loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
@@ -561,6 +562,14 @@ fn log(comptime message: []const u8) void {
|
|||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||||
|
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||||
|
/// `log` directly and always show, so a failed boot still explains itself.
|
||||||
|
fn progress(comptime message: []const u8) void {
|
||||||
|
if (!build_options.serial) return;
|
||||||
|
log(message);
|
||||||
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
|
|||||||
@@ -58,7 +58,6 @@ fn addUserBinary(
|
|||||||
b: *std.Build,
|
b: *std.Build,
|
||||||
target: std.Build.ResolvedTarget,
|
target: std.Build.ResolvedTarget,
|
||||||
runtime_module: *std.Build.Module,
|
runtime_module: *std.Build.Module,
|
||||||
posix_module: *std.Build.Module,
|
|
||||||
mmio_module: *std.Build.Module,
|
mmio_module: *std.Build.Module,
|
||||||
xkeyboard_config_module: *std.Build.Module,
|
xkeyboard_config_module: *std.Build.Module,
|
||||||
acpi_ids_module: *std.Build.Module,
|
acpi_ids_module: *std.Build.Module,
|
||||||
@@ -78,9 +77,6 @@ fn addUserBinary(
|
|||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "runtime", .module = runtime_module },
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
// POSIX/C compatibility layer, available to any program that wants it
|
|
||||||
// (danos-native code uses `runtime` directly). See library/posix/.
|
|
||||||
.{ .name = "posix", .module = posix_module },
|
|
||||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||||
.{ .name = "mmio", .module = mmio_module },
|
.{ .name = "mmio", .module = mmio_module },
|
||||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||||
@@ -100,6 +96,105 @@ fn addUserBinary(
|
|||||||
return exe;
|
return exe;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||||
|
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||||
|
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||||
|
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||||
|
const KernelModules = struct {
|
||||||
|
boot_handoff: *std.Build.Module,
|
||||||
|
abi: *std.Build.Module,
|
||||||
|
device_abi: *std.Build.Module,
|
||||||
|
architecture: *std.Build.Module,
|
||||||
|
platform: *std.Build.Module,
|
||||||
|
parameters: *std.Build.Module,
|
||||||
|
initial_ramdisk: *std.Build.Module,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||||
|
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||||
|
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||||
|
/// build option baked into `build_options`.
|
||||||
|
fn addKernel(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_target: std.Build.ResolvedTarget,
|
||||||
|
optimize: std.builtin.OptimizeMode,
|
||||||
|
modules: KernelModules,
|
||||||
|
test_case: ?[]const u8,
|
||||||
|
serial: bool,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||||
|
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||||
|
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||||
|
const build_options = b.addOptions();
|
||||||
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
|
build_options.addOption(bool, "serial", serial);
|
||||||
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = "kernel",
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||||
|
.target = kernel_target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
|
.stack_protector = false,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||||
|
.{ .name = "abi", .module = modules.abi },
|
||||||
|
.{ .name = "device-abi", .module = modules.device_abi },
|
||||||
|
.{ .name = "architecture", .module = modules.architecture },
|
||||||
|
.{ .name = "platform", .module = modules.platform },
|
||||||
|
.{ .name = "parameters", .module = modules.parameters },
|
||||||
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
|
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||||
|
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||||
|
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||||
|
/// bundle its own kernel while sharing the (serial-independent) loader, init, and
|
||||||
|
/// ramdisk. Returns the image's LazyPath.
|
||||||
|
fn addBootImage(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_bin: std.Build.LazyPath,
|
||||||
|
efi_bin: std.Build.LazyPath,
|
||||||
|
init_bin: std.Build.LazyPath,
|
||||||
|
initial_ramdisk_img: std.Build.LazyPath,
|
||||||
|
) std.Build.LazyPath {
|
||||||
|
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||||
|
mk_fat.addArg("64"); // MiB
|
||||||
|
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||||
|
mk_fat.addFileArg(efi_bin);
|
||||||
|
mk_fat.addArg("system/kernel");
|
||||||
|
mk_fat.addFileArg(kernel_bin);
|
||||||
|
mk_fat.addArg("system/services/init");
|
||||||
|
mk_fat.addFileArg(init_bin);
|
||||||
|
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||||
|
mk_fat.addFileArg(initial_ramdisk_img);
|
||||||
|
return fat_image;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
@@ -132,8 +227,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||||
// nodes. Also shared reference data.
|
// nodes. Also shared reference data.
|
||||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||||
// same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig,
|
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||||
// no kernel imports — one source, two builds.
|
// Pure Zig, no kernel imports — one source, two builds.
|
||||||
const aml_module = b.addModule("aml", .{
|
const aml_module = b.addModule("aml", .{
|
||||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||||
});
|
});
|
||||||
@@ -142,6 +237,28 @@ pub fn build(b: *std.Build) void {
|
|||||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||||
|
// class requests, descriptors) and the USB class-code taxonomy — the flat
|
||||||
|
// reference the xHCI bus driver, the USB class drivers, and the device
|
||||||
|
// manager's identity matcher all share. Pure data, like pci-class/acpi-ids.
|
||||||
|
const usb_abi_module = b.addModule("usb-abi", .{
|
||||||
|
.root_source_file = b.path("system/devices/usb-abi.zig"),
|
||||||
|
});
|
||||||
|
const usb_ids_module = b.addModule("usb-ids", .{
|
||||||
|
.root_source_file = b.path("system/devices/usb-ids.zig"),
|
||||||
|
});
|
||||||
|
// The USB transfer protocol: what a USB class driver says to the xHCI bus
|
||||||
|
// driver to drive its device (open / control / interrupt / bulk). A protocol
|
||||||
|
// module like vfs-protocol, shared by the bus driver and every class driver.
|
||||||
|
const usb_transfer_protocol_module = b.addModule("usb-transfer-protocol", .{
|
||||||
|
.root_source_file = b.path("system/drivers/usb-xhci-bus/usb-transfer-protocol.zig"),
|
||||||
|
});
|
||||||
|
// The block-device protocol: read/write of fixed-size blocks, spoken between a
|
||||||
|
// filesystem and a block driver (usb-storage). A protocol module like the rest.
|
||||||
|
const block_protocol_module = b.addModule("block-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/block/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||||
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||||
// in one place instead of scattered across the tree. See system/parameters.zig.
|
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||||
@@ -225,8 +342,27 @@ pub fn build(b: *std.Build) void {
|
|||||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||||
});
|
});
|
||||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||||
|
// The USB transfer protocol, so runtime.usb (the class-driver client) can speak
|
||||||
|
// it, the way runtime.input speaks the input protocol.
|
||||||
|
runtime_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||||
|
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||||
|
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||||
|
|
||||||
// The power protocol: system power's domain-named surface (docs/m21-plan.md).
|
// The display protocol, so runtime.display (the compositor client) and the display
|
||||||
|
// service both speak it through the runtime, like the other protocol modules.
|
||||||
|
const display_protocol_module = b.addModule("display-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/display/protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||||
|
|
||||||
|
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||||
|
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||||
|
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||||
|
|
||||||
|
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||||
const power_protocol_module = b.addModule("power-protocol", .{
|
const power_protocol_module = b.addModule("power-protocol", .{
|
||||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||||
});
|
});
|
||||||
@@ -253,18 +389,6 @@ pub fn build(b: *std.Build) void {
|
|||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
// The POSIX / C compatibility layer, a separate library layered strictly over the
|
|
||||||
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
|
|
||||||
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
|
|
||||||
// and library/posix/posix.zig.
|
|
||||||
const posix_module = b.addModule("posix", .{
|
|
||||||
.root_source_file = b.path("library/posix/posix.zig"),
|
|
||||||
.imports = &.{
|
|
||||||
.{ .name = "runtime", .module = runtime_module },
|
|
||||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
|
||||||
},
|
|
||||||
});
|
|
||||||
|
|
||||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||||
@@ -274,9 +398,12 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
// The serial-console log sink. Off by default: a real machine often has no
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||||
const build_options_module = build_options.createModule();
|
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||||
|
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||||
|
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||||
|
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -288,41 +415,17 @@ pub fn build(b: *std.Build) void {
|
|||||||
.abi = .none,
|
.abi = .none,
|
||||||
});
|
});
|
||||||
|
|
||||||
const exe = b.addExecutable(.{
|
const kernel_modules = KernelModules{
|
||||||
.name = "kernel",
|
.boot_handoff = boot_handoff_module,
|
||||||
.root_module = b.createModule(.{
|
.abi = abi_module,
|
||||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
.device_abi = device_abi_module,
|
||||||
.target = kernel_target,
|
.architecture = architecture_module,
|
||||||
.optimize = optimize,
|
.platform = platform_module,
|
||||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
.parameters = parameters_module,
|
||||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
.initial_ramdisk = initial_ramdisk_module,
|
||||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
};
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||||
.stack_protector = false,
|
|
||||||
.imports = &.{
|
|
||||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
|
||||||
.{ .name = "abi", .module = abi_module },
|
|
||||||
.{ .name = "device-abi", .module = device_abi_module },
|
|
||||||
.{ .name = "architecture", .module = architecture_module },
|
|
||||||
.{ .name = "platform", .module = platform_module },
|
|
||||||
.{ .name = "parameters", .module = parameters_module },
|
|
||||||
.{ .name = "build_options", .module = build_options_module },
|
|
||||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
|
||||||
},
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
|
||||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
|
||||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
|
||||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
|
||||||
exe.use_llvm = true;
|
|
||||||
exe.use_lld = true;
|
|
||||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
|
||||||
// linker's AT() clauses give each segment a low physical load address
|
|
||||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
|
||||||
exe.image_base = 0xFFFFFFFF80100000;
|
|
||||||
|
|
||||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
@@ -336,7 +439,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||||
// started in ring 3 by the kernel's user-ELF loader.
|
// started in ring 3 by the kernel's user-ELF loader.
|
||||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||||
b.getInstallStep().dependOn(&init_install.step);
|
b.getInstallStep().dependOn(&init_install.step);
|
||||||
|
|
||||||
@@ -344,24 +447,48 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Each is built by the same user-binary recipe, then packed into one image by
|
// Each is built by the same user-binary recipe, then packed into one image by
|
||||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
// usb-ids.packTriple.
|
||||||
|
usb_xhci_bus_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||||
|
usb_xhci_bus_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||||
|
usb_xhci_bus_exe.root_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||||
|
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||||
|
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||||
|
// the input service. They build chapter-9 class requests from usb-abi.
|
||||||
|
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||||
|
usb_hid_keyboard_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||||
|
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||||
|
usb_hid_mouse_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||||
|
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||||
|
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||||
|
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||||
|
usb_storage_exe.root_module.addImport("block-protocol", block_protocol_module);
|
||||||
|
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||||
|
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||||
|
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||||
|
const display_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||||
|
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||||
|
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||||
|
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||||
|
const shm_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-client", "system/services/shm-client/shm-client.zig");
|
||||||
|
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||||
|
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||||
// The PCI bus driver decodes each function's class triple to human names in its
|
// The PCI bus driver decodes each function's class triple to human names in its
|
||||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||||
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||||
// what the driver-restart scenario drives the crash-loop cap with.
|
// what the driver-restart scenario drives the crash-loop cap with.
|
||||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||||
// The discovery service: one swappable process per firmware
|
// The discovery service: one swappable process per firmware
|
||||||
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||||
// "discovery" so the device manager never learns which firmware it is on.
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||||
// flattened device tree — the aarch64 target flips the default when it
|
// flattened device tree — the aarch64 target flips the default when it
|
||||||
@@ -373,18 +500,22 @@ pub fn build(b: *std.Build) void {
|
|||||||
.acpi => "system/services/acpi/acpi.zig",
|
.acpi => "system/services/acpi/acpi.zig",
|
||||||
.fdt => "system/services/fdt/fdt.zig",
|
.fdt => "system/services/fdt/fdt.zig",
|
||||||
};
|
};
|
||||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||||
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||||
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||||
|
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||||
|
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||||
|
device_manager_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||||
|
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||||
|
|
||||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
@@ -396,10 +527,6 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||||
mk_run.addArg("vfs-test");
|
mk_run.addArg("vfs-test");
|
||||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||||
mk_run.addArg("hpet");
|
|
||||||
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
|
||||||
mk_run.addArg("bus");
|
|
||||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
|
||||||
mk_run.addArg("ps2-bus");
|
mk_run.addArg("ps2-bus");
|
||||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||||
mk_run.addArg("ps2-keyboard");
|
mk_run.addArg("ps2-keyboard");
|
||||||
@@ -408,6 +535,26 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||||
mk_run.addArg("usb-xhci-bus");
|
mk_run.addArg("usb-xhci-bus");
|
||||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-hid-keyboard");
|
||||||
|
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-hid-mouse");
|
||||||
|
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-storage");
|
||||||
|
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("fat");
|
||||||
|
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("fat-test");
|
||||||
|
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("display");
|
||||||
|
mk_run.addFileArg(display_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("display-demo");
|
||||||
|
mk_run.addFileArg(display_demo_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("virtio-gpu");
|
||||||
|
mk_run.addFileArg(virtio_gpu_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("shm-server");
|
||||||
|
mk_run.addFileArg(shm_server_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("shm-client");
|
||||||
|
mk_run.addFileArg(shm_client_exe.getEmittedBin());
|
||||||
mk_run.addArg("pci-bus");
|
mk_run.addArg("pci-bus");
|
||||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||||
mk_run.addArg("crash-test");
|
mk_run.addArg("crash-test");
|
||||||
@@ -428,6 +575,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||||
mk_run.addArg("process-test");
|
mk_run.addArg("process-test");
|
||||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("log-flush");
|
||||||
|
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||||
|
|
||||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||||
@@ -435,12 +584,16 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ vfs_exe, "system/services" },
|
.{ vfs_exe, "system/services" },
|
||||||
.{ device_manager_exe, "system/services" },
|
.{ device_manager_exe, "system/services" },
|
||||||
.{ input_exe, "system/services" },
|
.{ input_exe, "system/services" },
|
||||||
.{ hpet_exe, "system/drivers" },
|
|
||||||
.{ bus_exe, "system/drivers" },
|
|
||||||
.{ ps2_bus_exe, "system/drivers" },
|
.{ ps2_bus_exe, "system/drivers" },
|
||||||
.{ ps2_keyboard_exe, "system/drivers" },
|
.{ ps2_keyboard_exe, "system/drivers" },
|
||||||
.{ ps2_mouse_exe, "system/drivers" },
|
.{ ps2_mouse_exe, "system/drivers" },
|
||||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||||
|
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||||
|
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||||
|
.{ usb_storage_exe, "system/drivers" },
|
||||||
|
.{ fat_exe, "system/services" },
|
||||||
|
.{ display_exe, "system/services" },
|
||||||
|
.{ log_flush_exe, "system/services" },
|
||||||
}) |entry| {
|
}) |entry| {
|
||||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||||
b.getInstallStep().dependOn(&step.step);
|
b.getInstallStep().dependOn(&step.step);
|
||||||
@@ -453,6 +606,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||||
// Each is its own binary/entry (a loader is built for its own target); today
|
// Each is its own binary/entry (a loader is built for its own target); today
|
||||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||||
|
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||||
|
// which firmware may mirror to a serial console) are silenced by default — a
|
||||||
|
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||||
|
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||||
|
const loader_options = b.addOptions();
|
||||||
|
loader_options.addOption(bool, "serial", serial);
|
||||||
|
const loader_options_module = loader_options.createModule();
|
||||||
const efiexe = b.addExecutable(.{
|
const efiexe = b.addExecutable(.{
|
||||||
.name = "BOOTX64",
|
.name = "BOOTX64",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
@@ -465,6 +625,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.imports = &.{
|
.imports = &.{
|
||||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
|
.{ .name = "build_options", .module = loader_options_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
@@ -474,6 +635,32 @@ pub fn build(b: *std.Build) void {
|
|||||||
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||||
b.getInstallStep().dependOn(&efi_install.step);
|
b.getInstallStep().dependOn(&efi_install.step);
|
||||||
|
|
||||||
|
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||||
|
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||||
|
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||||
|
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||||
|
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||||
|
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||||
|
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||||
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|
||||||
|
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||||
|
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||||
|
// log captured to serial0 — without baking serial into the image users flash.
|
||||||
|
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||||
|
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||||
|
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||||
|
|
||||||
|
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||||
|
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||||
|
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
check_fat.addArg("--verify");
|
||||||
|
check_fat.addFileArg(fat_image);
|
||||||
|
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||||
|
check_fat_step.dependOn(&check_fat.step);
|
||||||
|
|
||||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
// Firmware lives in different places per OS/distro, so probe the known
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
@@ -534,10 +721,14 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
run_efi.addArg("-drive");
|
run_efi.addArg("-drive");
|
||||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
// Present the FHS zig-out to the guest as a FAT drive — it is the boot volume.
|
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||||
|
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||||
|
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||||
|
run_efi.addArg("-drive");
|
||||||
|
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||||
run_efi.addArgs(&.{
|
run_efi.addArgs(&.{
|
||||||
"-drive",
|
"-device",
|
||||||
b.fmt("format=raw,file=fat:rw:{s}", .{b.install_path}),
|
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||||
"-net",
|
"-net",
|
||||||
"none",
|
"none",
|
||||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||||
@@ -556,8 +747,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||||
run_efi.step.dependOn(b.getInstallStep());
|
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||||
|
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||||
|
// scratch dir first.
|
||||||
run_efi.step.dependOn(&make_log_dir.step);
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
@@ -589,6 +782,17 @@ pub fn build(b: *std.Build) void {
|
|||||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||||
|
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||||
|
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||||
|
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||||
|
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||||
|
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||||
|
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||||
|
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||||
|
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||||
|
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||||
|
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||||
|
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||||
}) |root| {
|
}) |root| {
|
||||||
const mod_tests = b.addTest(.{
|
const mod_tests = b.addTest(.{
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
@@ -615,6 +819,21 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||||
|
|
||||||
|
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||||
|
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||||
|
// loop above.
|
||||||
|
const time_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("library/runtime/time.zig"),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||||
|
|
||||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||||
|
|||||||
+45
-15
@@ -69,7 +69,18 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||||
service layered on top.
|
service layered on top.
|
||||||
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
19. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||||
|
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||||
|
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||||
|
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||||
|
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||||
|
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||||
|
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||||
|
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||||
|
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||||
|
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||||
|
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||||
|
20. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
@@ -82,9 +93,19 @@ Start with the north star:
|
|||||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||||
fault isolation + live restart — the reincarnation-server + capability model that
|
fault isolation + live restart — the reincarnation-server + capability model that
|
||||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||||
|
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||||
|
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||||
|
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||||
|
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||||
|
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||||
|
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||||
|
|
||||||
Cutting across all of these:
|
Cutting across all of these:
|
||||||
|
|
||||||
|
- **[system-requirements.md](system-requirements.md) — system requirements.** The
|
||||||
|
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||||
|
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||||
|
plain-language guide matching Intel/AMD CPU generations by name.
|
||||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||||
@@ -96,7 +117,16 @@ Cutting across all of these:
|
|||||||
when to build it, and how to keep it architecture-agnostic.
|
when to build it, and how to keep it architecture-agnostic.
|
||||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs.
|
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||||
|
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||||
|
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||||
|
service: button/lid/battery events published to subscribers, and init's orderly
|
||||||
|
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||||
|
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||||
|
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||||
|
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||||
|
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||||
|
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||||
right choice depends on whether danos is chasing real-time or resilience.
|
right choice depends on whether danos is chasing real-time or resilience.
|
||||||
@@ -153,7 +183,7 @@ addressed as **`system/services/init`** — the repeated leaf resolves away:
|
|||||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||||
|----------------------------------------|--------------------------------------------|
|
|----------------------------------------|--------------------------------------------|
|
||||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||||
|
|
||||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||||
@@ -178,21 +208,22 @@ system/ → /system danos's own internals (the self-representation)
|
|||||||
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||||
vfs.zig, vfs-test.zig, protocol.zig)
|
vfs.zig, vfs-test.zig, protocol.zig)
|
||||||
library/ → /lib libraries, one sub-directory each
|
library/ → /lib libraries, one sub-directory each
|
||||||
runtime/ the danos-native runtime — the stable application ABI
|
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||||
posix/ POSIX/C compatibility, layered over runtime
|
|
||||||
boot/ → /boot the loaders
|
boot/ → /boot the loaders
|
||||||
tools/ test/ host-side build + QEMU test harness
|
tools/ test/ host-side build + QEMU test harness
|
||||||
```
|
```
|
||||||
|
|
||||||
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the POSIX
|
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
|
||||||
layer imports by name. `usb`/`block` drivers will expose their protocols the same way.
|
file API (`runtime.fs`) imports by name. `usb`/`block` drivers expose their protocols the
|
||||||
|
same way.
|
||||||
|
|
||||||
`library/posix/` is special: it is the **one place** POSIX/C spellings are allowed
|
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||||
verbatim (`stat`, `O_CREAT`, `fopen`, `errno`). Everywhere else follows the danos
|
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||||
naming rule with no exception — see [coding-standards.md](coding-standards.md). The
|
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||||
POSIX layer calls the runtime, never the kernel's system calls directly, so it never
|
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||||
appears in the private-ABI path.
|
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||||
|
exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||||
|
|
||||||
## Source map
|
## Source map
|
||||||
|
|
||||||
@@ -216,9 +247,8 @@ appears in the private-ABI path.
|
|||||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||||
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
|
||||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||||
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
+55
-1
@@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a
|
|||||||
part that answers "where are the tables?" — everything past the RSDP is just following
|
part that answers "where are the tables?" — everything past the RSDP is just following
|
||||||
more pointers the tables themselves provide.
|
more pointers the tables themselves provide.
|
||||||
|
|
||||||
|
## ACPI events: the SCI, the power button, and GPEs (M21)
|
||||||
|
|
||||||
|
The tables above are static description; ACPI is also a *live* channel. Hardware
|
||||||
|
raises the **SCI** (System Control Interrupt) — one shared, level-triggered line
|
||||||
|
whose vector the FADT names — and the OS reads status registers to learn what
|
||||||
|
happened: a fixed event like the power button, or a **General-Purpose Event**
|
||||||
|
(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML
|
||||||
|
to ring 3, the event side lives there too, in the same **acpi service** — the
|
||||||
|
device discoverer and the event source are one process, because both need the
|
||||||
|
namespace and the port grant.
|
||||||
|
|
||||||
|
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||||
|
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||||
|
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||||
|
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||||
|
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||||
|
header, while the blob resources are header-stripped bytecode that starts with
|
||||||
|
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||||
|
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||||
|
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||||
|
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||||
|
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||||
|
finds the line to `irq_bind`.
|
||||||
|
|
||||||
|
With those in hand the service enables ACPI mode (only if `SCI_EN` is clear —
|
||||||
|
some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||||
|
|
||||||
|
- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status.
|
||||||
|
The handler clears it (write-1-to-clear), logs the press, and publishes a
|
||||||
|
[`power`](power.md) `power_button` event to subscribers.
|
||||||
|
- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the
|
||||||
|
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||||
|
method, drains the **Notify** queue that method produced, maps each notified
|
||||||
|
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||||
|
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||||
|
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||||
|
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||||
|
per evaluation; everything else a handler needs (field access, control flow,
|
||||||
|
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||||
|
|
||||||
|
**How this is tested.** QEMU cannot raise GPEs deterministically on this config,
|
||||||
|
so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||||
|
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||||
|
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||||
|
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||||
|
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||||
|
are interface-complete but validated on real hardware later.
|
||||||
|
|
||||||
|
The service surface these events are *published on* — subscription, the event
|
||||||
|
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||||
|
|
||||||
## Related
|
## Related
|
||||||
|
|
||||||
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
||||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
||||||
the ACPI-reclaim memory the RSDP lives in.
|
the ACPI-reclaim memory the RSDP lives in.
|
||||||
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
||||||
tables into one neutral device model shared with the ARM device-tree path.
|
tables into one neutral device model shared with the ARM device-tree path, and how
|
||||||
|
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||||
|
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||||
|
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||||
module and never names ACPI directly.
|
module and never names ACPI directly.
|
||||||
|
|||||||
+14
-11
@@ -65,15 +65,18 @@ Three, and only three.
|
|||||||
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||||
`fwrite` any more.
|
`fwrite` any more.
|
||||||
|
|
||||||
**This exception is scoped to one place: `library/posix/`.** A file under
|
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||||
`library/posix/` *is* the foreign ABI, so it keeps the ABI's spellings — that is the
|
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||||
whole rule for that directory. **Everywhere else, Zig/danos naming applies with no
|
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||||
POSIX exception**, so there is nothing to get wrong: if you're not in
|
layer is premature until danos actually needs it (see
|
||||||
`library/posix/`, expand it. A concept POSIX also has gets a danos name outside that
|
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||||
layer — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||||
flag, not `O_CREAT`; `library/posix/` is what maps `stat`→`status` and
|
`system` interface, so it keeps `open`/`read`/`errno`/`O_CREAT`. **Everywhere else,
|
||||||
`O_CREAT`→`create` at the boundary. (The `syscall` *wrappers* elsewhere are not an
|
Zig/danos naming applies with no exception**: a concept POSIX also has gets a danos
|
||||||
exception to this — they wrap the private danos ABI, so they use danos names.)
|
name — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||||
|
flag, not `O_CREAT`; the boundary is where `stat`→`status` and `O_CREAT`→`create` get
|
||||||
|
mapped. (The `syscall` *wrappers* elsewhere are not an exception — they wrap the
|
||||||
|
private danos ABI, so they use danos names.)
|
||||||
|
|
||||||
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||||
not ours, and are left alone:
|
not ours, and are left alone:
|
||||||
@@ -96,7 +99,7 @@ That's all — no Unix-abbreviation exception. The source directories are full w
|
|||||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||||
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||||
already tells you.
|
already tells you.
|
||||||
|
|
||||||
## A note on collisions
|
## A note on collisions
|
||||||
@@ -138,7 +141,7 @@ single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig
|
|||||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||||
|
|
||||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||||
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||||
resolving away. See the repository-layout section of [README.md](README.md).
|
resolving away. See the repository-layout section of [README.md](README.md).
|
||||||
|
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
|||||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||||
| /system/kernel | the kernel image |
|
| /system/kernel | the kernel image |
|
||||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||||
@@ -61,8 +61,8 @@ to the driver in the order written, and a read consumes what is there. Terminals
|
|||||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||||
that program, minus the client half.
|
is already that program, minus the file-node client half.
|
||||||
|
|
||||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||||
|
|||||||
@@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea
|
|||||||
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
||||||
timer — a later step.
|
timer — a later step.
|
||||||
|
|
||||||
|
### Is the TSC trustworthy? Invariant, and synchronized
|
||||||
|
|
||||||
|
A cycle counter is only a valid *clock* if two things hold, and danos checks both,
|
||||||
|
because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET.
|
||||||
|
|
||||||
|
**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with
|
||||||
|
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||||
|
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||||
|
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||||
|
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||||
|
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||||
|
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||||
|
measured frequency without the invariance guarantee is not enough.
|
||||||
|
|
||||||
|
**Synchronized.** Each core has its own TSC. Even invariant ones can start at different
|
||||||
|
values (a second socket, some firmware), so a thread migrating from a core reading
|
||||||
|
`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp
|
||||||
|
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||||
|
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||||
|
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||||
|
come up one at a time ([smp.md](smp.md)).
|
||||||
|
|
||||||
|
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||||
|
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||||
|
**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor
|
||||||
|
drift with frequency. It costs a memory-mapped read instead of a register read, but it
|
||||||
|
keeps time *accurate*, which is the whole point. The switch preserves the current value,
|
||||||
|
so the clock never jumps. The boot log names the outcome:
|
||||||
|
|
||||||
|
```
|
||||||
|
/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD
|
||||||
|
/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG)
|
||||||
|
```
|
||||||
|
|
||||||
## Two kinds of vector, one dispatch
|
## Two kinds of vector, one dispatch
|
||||||
|
|
||||||
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
||||||
|
|||||||
+21
-5
@@ -51,7 +51,16 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
|||||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||||
depends on when it lands.
|
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||||
|
|
||||||
|
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||||
|
identical (parent, class, identity, resources) tuple returns the existing id
|
||||||
|
instead of appending a duplicate. The kernel table has no unregister, so without
|
||||||
|
this a restarted registering bus would re-report its children as fresh nodes on
|
||||||
|
every respawn. Idempotence is what makes restart-and-re-report sound for *every*
|
||||||
|
reporting bus — pci-bus, the acpi service, a future fdt service — not just one,
|
||||||
|
and it is why supervision (below) can prune a dead bus's subtree and trust the
|
||||||
|
restarted instance to rebuild exactly the same ids.
|
||||||
|
|
||||||
## The protocol
|
## The protocol
|
||||||
|
|
||||||
@@ -136,10 +145,17 @@ published exit events, signals + `runtime.process`). On top of those:
|
|||||||
the mouse and keyboard QEMU already hangs off it.
|
the mouse and keyboard QEMU already hangs off it.
|
||||||
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||||
to a manager-internal seam.
|
to a manager-internal seam.
|
||||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19)
|
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||||
then the acpi service (M20) moved enumeration to ring 3; the kernel seeds
|
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||||
only the host bridge and the acpi-tables node. See
|
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||||
[m19-m20-plan.md](m19-m20-plan.md).
|
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||||
|
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||||
|
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||||
|
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||||
|
PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3
|
||||||
|
— each in a single phase so no device is ever matched from both sources at
|
||||||
|
once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus
|
||||||
|
already reports PCI functions (M20.2).
|
||||||
|
|
||||||
## Settled questions (2026-07-12)
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
|||||||
@@ -191,3 +191,57 @@ and registers + reports each `_HID` device — the device manager matches driver
|
|||||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||||
entirely in user space; the kernel seeds only the host bridge and the
|
entirely in user space; the kernel seeds only the host bridge and the
|
||||||
acpi-tables node.
|
acpi-tables node.
|
||||||
|
|
||||||
|
## Discovery is a swappable process per firmware (M19–M20)
|
||||||
|
|
||||||
|
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||||
|
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||||
|
to do it before the second architecture rather than after. Everything at and
|
||||||
|
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||||
|
containment, reports, matching, supervision — is generic and may never become
|
||||||
|
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||||
|
isolated as **one swappable process per firmware**:
|
||||||
|
|
||||||
|
- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi
|
||||||
|
service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML.
|
||||||
|
- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is
|
||||||
|
an **fdt service**: it claims a `devicetree-blob` node and walks the tree —
|
||||||
|
pure data, no bytecode, so it needs neither a port grant nor an interpreter,
|
||||||
|
strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md)
|
||||||
|
bring-up fills it in.)
|
||||||
|
|
||||||
|
The device manager spawns the discoverer under the **neutral ramdisk name
|
||||||
|
`discovery`** and never learns which firmware it is on; the build's
|
||||||
|
`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the
|
||||||
|
aarch64 target flips the default when it lands). The manager owns the device
|
||||||
|
tree as *data* and touches no hardware, ever — firmware bytecode runs only
|
||||||
|
inside the crashable, supervised discoverer, so an AML fault can never take
|
||||||
|
down the supervisor.
|
||||||
|
|
||||||
|
Two consequences of neutrality bind on later work:
|
||||||
|
|
||||||
|
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||||
|
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||||
|
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||||
|
`ServiceId.power`, and subscribers never learn the difference.
|
||||||
|
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||||
|
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||||
|
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||||
|
report a real node.
|
||||||
|
|
||||||
|
Two supporting decisions keep the kernel's remaining slice honest:
|
||||||
|
|
||||||
|
- **The AML interpreter is a shared build module**, compiled into both the
|
||||||
|
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||||
|
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||||
|
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||||
|
device count across the ring-3 move.
|
||||||
|
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||||
|
PCI functions carry BAR resources, and `device_register` containment demands
|
||||||
|
the bridge own windows that cover them. Those apertures are derived
|
||||||
|
kernel-side from the boot memory map's MMIO holes (regions that are neither
|
||||||
|
RAM nor tables) — mechanical, AML-free, and available at boot regardless of
|
||||||
|
what later moved to user space. The acpi service's authority is likewise
|
||||||
|
exactly one node: the `acpi-tables` node, whose broad io_port grant is the
|
||||||
|
documented trust boundary for the one process allowed to run firmware
|
||||||
|
bytecode.
|
||||||
|
|||||||
@@ -0,0 +1,181 @@
|
|||||||
|
# Display service — build plan (v1: the dumb-framebuffer compositor)
|
||||||
|
|
||||||
|
The ordered, checkpointable build-out for [display.md](display.md). Each milestone is
|
||||||
|
small, lands on its own, and ends in a **verifiable gate** — shaped for a `/loop` run.
|
||||||
|
Read [display.md](display.md) first for the *why*; this is the *what* and the *order*.
|
||||||
|
|
||||||
|
## Locked decisions (do not relitigate)
|
||||||
|
|
||||||
|
- **Handoff = device node + write-combining `mmio_map`.** The kernel seeds a synthetic
|
||||||
|
`display0` node from `BootInformation.framebuffer`; the service claims + WC-maps it.
|
||||||
|
(Not a bespoke `framebuffer_map` syscall — the device route inherits ownership,
|
||||||
|
release-on-death, and re-claim-on-restart.)
|
||||||
|
- **v1 = the full compositor pipeline on the dumb framebuffer.** One `display` service
|
||||||
|
owns the LFB + a cacheable back buffer + a layer stack; double-buffer + damage-driven
|
||||||
|
present; clients draw via server-side commands. **No** runtime mode-setting, **no**
|
||||||
|
shared-memory surfaces — both deferred (see display.md, "What v1 does not do").
|
||||||
|
|
||||||
|
## Conventions
|
||||||
|
|
||||||
|
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||||
|
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||||
|
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||||
|
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||||
|
`runtime` module.
|
||||||
|
|
||||||
|
## How to verify along the way
|
||||||
|
|
||||||
|
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||||
|
pitch/format blits are all host-testable with a fake framebuffer).
|
||||||
|
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||||
|
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||||
|
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||||
|
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## D1 — The handoff primitive (kernel) ✅
|
||||||
|
|
||||||
|
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||||
|
|
||||||
|
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||||
|
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||||
|
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||||
|
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||||
|
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||||
|
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||||
|
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||||
|
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||||
|
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||||
|
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||||
|
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||||
|
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||||
|
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||||
|
|
||||||
|
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||||
|
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||||
|
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||||
|
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||||
|
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||||
|
screenshot-of-a-fill gate because it proves the *actual* WC property headlessly; the
|
||||||
|
visible fill folds into D2's gate (the service clears the screen through the back buffer).
|
||||||
|
Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `device-list`,
|
||||||
|
`device-manager` all still pass with the +1 device in the table.
|
||||||
|
|
||||||
|
## D2 — Service skeleton, protocol, runtime module ✅
|
||||||
|
|
||||||
|
Stand up the named service and the double-buffer, no layers yet.
|
||||||
|
|
||||||
|
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||||
|
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||||
|
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||||
|
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||||
|
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||||
|
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||||
|
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||||
|
until D3. Init clears the back buffer and presents it — the double-buffer path.
|
||||||
|
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||||
|
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||||
|
lookup with retry (model: block.zig).
|
||||||
|
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||||
|
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||||
|
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||||
|
`/system/services/display`.
|
||||||
|
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||||
|
a fixed kernel-stack `frames` array. Rewrote `systemMmap` to map page-by-page with
|
||||||
|
rollback (no scratch array) and raised the cap to 8192 pages (32 MiB) — enough for a
|
||||||
|
4K back buffer. A real limitation met, exactly the kind this project chases.
|
||||||
|
|
||||||
|
**Gate (met, automated):** `python3 test/qemu_test.py display-service` spawns the
|
||||||
|
compositor and matches its own serial heartbeats — `display: online {w}x{h} pitch …`
|
||||||
|
followed by `display: presented frame 0` — which it prints only after the whole
|
||||||
|
claim → WC-map → back-buffer → clear → present chain succeeds (matched on serial like the
|
||||||
|
fault cases, since a lone blocking service can't reschedule the in-kernel test context to
|
||||||
|
poll). Regression-checked: `usermem`, `heap` (the `mmap` rewrite), `init` (the boot-list
|
||||||
|
addition), and D1's `display` all still pass.
|
||||||
|
|
||||||
|
## D3 — Layer stack + compositor + damage present ✅
|
||||||
|
|
||||||
|
The heart: composite an ordered layer stack, present only what changed.
|
||||||
|
|
||||||
|
- [x] A layer table (16 slots): each `Layer` = position, z, visible, a server-owned
|
||||||
|
`mmap`'d surface (freed on `destroy_layer`). `damage` accumulates the dirty screen
|
||||||
|
region since the last present.
|
||||||
|
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||||
|
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||||
|
`damage`, `present`.
|
||||||
|
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||||
|
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||||
|
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||||
|
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||||
|
Colour packing (rgbx/bgrx) is `protocol.pack`, also host-tested.
|
||||||
|
- [x] Host tests (`zig build test`, green): rect intersect/unite, `fillRect` clipping +
|
||||||
|
`stride > width` padding, `composite` overlap-shows-top + damage clipping, `blitTile`
|
||||||
|
unaligned read + clipping, and `pack` for both pixel formats.
|
||||||
|
|
||||||
|
**Gate (met):** `zig build test` green for the compositor + pack unit tests, **and** the
|
||||||
|
`display-service` case's startup self-check composites two overlapping layers on the real
|
||||||
|
framebuffer and reads back the composited pixels — overlap = top layer, outside = bottom
|
||||||
|
layer — logging `display: compositor self-check ok` (matched by the harness).
|
||||||
|
|
||||||
|
## D4 — Client API + the demo client ✅
|
||||||
|
|
||||||
|
Prove the pipeline end-to-end from a separate process.
|
||||||
|
|
||||||
|
- [x] Finished [runtime/display.zig](../library/runtime/runtime.zig): a `Layer` handle with
|
||||||
|
`fill` / `blitTile` (inline tile) / `configure` (move/restack/show) / `damage` /
|
||||||
|
`destroy`, `createLayer`, and a `color(r,g,b)` helper (caches the mode, packs via
|
||||||
|
`protocol.pack`). Coordinates are signed over the wire (`@bitCast` both ways).
|
||||||
|
- [x] `system/services/display-demo/`: a hardware-free client (the `input-source` analog)
|
||||||
|
— a full-screen wallpaper layer, a rectangle that slides back and forth (moved by
|
||||||
|
`configure` each frame, so the compositor repaints old + new), and a cursor layer;
|
||||||
|
presents in a loop paced by `runtime.time`. Wired into build + initial-ramdisk.
|
||||||
|
- [x] **Bug this surfaced:** `protocol.message_maximum` was 4096, but the kernel caps
|
||||||
|
every IPC message at `MESSAGE_MAXIMUM` = 256 — so `replyWait` rejected the oversized
|
||||||
|
receive buffer with `-E2BIG` and the serve loop had been *spinning* since D2 (unseen,
|
||||||
|
as D2/D3 matched init-time heartbeats). Set it to 256; `blit_tile` is now explicitly
|
||||||
|
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shm surface.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py display-demo` spawns the service + `display-demo`;
|
||||||
|
the demo drives a run of frames of motion through the layer client API and logs
|
||||||
|
`display-demo: ok` (the visible motion is a screenshot via `zig build run-x86-64`).
|
||||||
|
Regression-checked: `zig build test`, `display` (D1), and `display-service` (D2/D3) all
|
||||||
|
still pass, and the default `zig build` is clean.
|
||||||
|
|
||||||
|
## D5 — Test cases + docs ✅
|
||||||
|
|
||||||
|
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||||
|
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||||
|
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||||
|
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||||
|
the pure host tests (`zig build test`).
|
||||||
|
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||||
|
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||||
|
memory marked DONE with the commits.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||||
|
`zig build test` is green, and the default `zig build` is clean.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## v1 status: complete
|
||||||
|
|
||||||
|
D1–D5 done. The display service is a working framebuffer compositor: it owns the
|
||||||
|
framebuffer (write-combining), composites a z-ordered layer stack into a cacheable back
|
||||||
|
buffer, presents only the damaged region, and is driven over IPC by the `runtime.display`
|
||||||
|
client — proven end-to-end by a separate demo process. Two limitations are deliberate and
|
||||||
|
documented (docs/display.md): no runtime mode-setting (native backend) and no true vsync
|
||||||
|
(no vblank on a dumb framebuffer). Next steps are the Deferred items below.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Deferred (explicitly not in this plan)
|
||||||
|
|
||||||
|
- **Shared-memory surfaces** — generalize M13 capability passing to memory objects
|
||||||
|
(`shm_create`/`shm_map`), so bitmap clients hand the compositor a rendered surface
|
||||||
|
instead of drawing commands. The compositor's layer model already anticipates it.
|
||||||
|
- **Native backend (Bochs DISPI, then virtio-gpu)** — behind the same internal backend
|
||||||
|
interface as the dumb framebuffer: EDID mode list + runtime resolution/bpp change +
|
||||||
|
(eventually) a vblank/flip path for true vsync.
|
||||||
|
- **Driver/compositor process split** — only when a second backend or a second head makes
|
||||||
|
the abstraction pay for itself.
|
||||||
@@ -0,0 +1,173 @@
|
|||||||
|
# Display v2 — build plan (pluggable scanout: GOP floor + virtio-gpu native)
|
||||||
|
|
||||||
|
The ordered, checkpointable build-out for [display-v2.md](display-v2.md). Each milestone
|
||||||
|
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||||
|
[display-plan.md](display-plan.md). Read display-v2.md first for the *why*.
|
||||||
|
|
||||||
|
## Locked decisions (do not relitigate)
|
||||||
|
|
||||||
|
- **First native backend = virtio-gpu** (VM standard: mode-set + present/flush + vsync).
|
||||||
|
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||||
|
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||||
|
ever," not a live fall-back after a reprogram.
|
||||||
|
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||||
|
client-surface path.
|
||||||
|
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||||
|
|
||||||
|
## Conventions
|
||||||
|
|
||||||
|
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||||
|
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||||
|
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||||
|
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||||
|
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||||
|
|
||||||
|
## How to verify along the way
|
||||||
|
|
||||||
|
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||||
|
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||||
|
own pixels back**: the scanout resource is CPU-visible RAM (shm-backed) and the back buffer
|
||||||
|
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||||
|
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||||
|
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||||
|
"it's on screen."
|
||||||
|
|
||||||
|
- `zig build test` — host unit tests (backend selection, virtio struct sizes/encodings,
|
||||||
|
pixel-check helpers).
|
||||||
|
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial markers.
|
||||||
|
The virtio cases boot with `-device virtio-gpu` (a per-case `qemu_extra`).
|
||||||
|
- `run-x86-64` renders to a window — for the human's own satisfaction, **not** a gate.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## V1 — The scanout backend seam (refactor, no behaviour change) ✅
|
||||||
|
|
||||||
|
Extract scanout from the compositor so today's path becomes one backend among future ones.
|
||||||
|
|
||||||
|
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||||
|
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||||
|
(`canModeSet`/`hasVsync`, both false for GOP).
|
||||||
|
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||||
|
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||||
|
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||||
|
framebuffer geometry left in the compositor core.
|
||||||
|
- [x] The selection decision is the pure `chooseKind(native_available)` (gop unless a
|
||||||
|
native driver announced), split from the syscall-bound `select()`/`Gop.init()`.
|
||||||
|
|
||||||
|
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||||
|
is the only backend), and `zig build test` stays green.
|
||||||
|
|
||||||
|
## V2 — The `shm` cross-process memory capability (kernel) ✅
|
||||||
|
|
||||||
|
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||||
|
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||||
|
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||||
|
handle, maps them into the caller's shm arena → returns vaddr + handle; `shm_map(cap)`
|
||||||
|
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||||
|
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||||
|
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||||
|
dispatch by kind, so an `ShmObject` rides an `ipc_call` `send_cap` exactly like an
|
||||||
|
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||||
|
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||||
|
frames — the object owns them.
|
||||||
|
- [x] `library/runtime/shm.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||||
|
`map(handle) -> ptr`.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py shm` — `shm-client` creates a region, writes a
|
||||||
|
pattern, and passes its capability to `shm-server` as an `ipc_call` send_cap; the server
|
||||||
|
`shm_map`s it and reads the **same bytes** back → `shm: shared 4096 bytes ok`. Guardrail:
|
||||||
|
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||||
|
tests all still pass — the handle-table change broke no existing IPC.
|
||||||
|
|
||||||
|
## V3 — The virtio-gpu driver: bring-up + a frame on screen ✅
|
||||||
|
|
||||||
|
- [x] `system/drivers/virtio-gpu/`: claim the virtio-gpu PCI function (device-manager
|
||||||
|
match on the display/other class triple, driver self-confirms vendor 0x1AF4/device
|
||||||
|
0x1050 from config space), enable memory-space + bus-master, walk the vendor
|
||||||
|
capabilities in config space to find common-config + notify, map the BAR, negotiate
|
||||||
|
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||||
|
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||||
|
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||||
|
shm-shared surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||||
|
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||||
|
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||||
|
|
||||||
|
**Gate (met):** the `virtio-gpu` case (QEMU `-device virtio-gpu-pci`) boots the
|
||||||
|
device-manager stack, which discovers the function and spawns the driver; the driver writes
|
||||||
|
a known test pattern into the scanout backing, `transfer_to_host_2d` + `resource_flush`es
|
||||||
|
it, and **waits for the device's used-ring ack**, then reads the backing back and checks the
|
||||||
|
pattern — logging `virtio-gpu: scanout 640x480 online` and `virtio-gpu: flush acked, pixel
|
||||||
|
check ok`. That proves virtqueue + resource + attach + set_scanout + transfer + flush end to
|
||||||
|
end without a screenshot (the used-ring ack is the device confirming it consumed the frame).
|
||||||
|
|
||||||
|
## V4 — The native backend + hot-attach ✅
|
||||||
|
|
||||||
|
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared `shm` scanout surface
|
||||||
|
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||||
|
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||||
|
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||||
|
- [x] The driver **announces** to `.display` after bring-up (looks it up with a bounded retry,
|
||||||
|
sends `attach_scanout` with the geometry + the shared surface as an `ipc_call` send_cap).
|
||||||
|
The compositor maps it, looks up `.scanout` itself (no need to pass the endpoint — the
|
||||||
|
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||||
|
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||||
|
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||||
|
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shm_physical` (a
|
||||||
|
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||||
|
|
||||||
|
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||||
|
boots the whole system) starts the compositor + `display-demo` + device-manager; the driver
|
||||||
|
announces, the compositor logs `display: scanout upgraded to virtio-gpu`, drives frames through
|
||||||
|
the native backend, and **reads a pixel back** from the shared surface after a present to
|
||||||
|
confirm the composited frame landed (`display: native present verified`), while `display-demo:
|
||||||
|
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||||
|
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||||
|
|
||||||
|
## V5 — Mode-setting, EDID, and vsync ✅
|
||||||
|
|
||||||
|
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||||
|
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||||
|
resource + shared surface are sized to the largest mode, so `set_mode` just re-points the
|
||||||
|
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||||
|
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||||
|
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||||
|
fence when the frame is on screen, which the used-ring ack the synchronous present waits on
|
||||||
|
already gates — a tear-free present.
|
||||||
|
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasVsync` = true.
|
||||||
|
|
||||||
|
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||||
|
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||||
|
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||||
|
fenced present path is exercised and confirmed (`display: vsync present ok`) — both from serial,
|
||||||
|
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||||
|
|
||||||
|
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||||
|
|
||||||
|
- [x] The virtio-gpu driver now **hellos** the device manager (role: bus) so it is properly
|
||||||
|
supervised — no longer stopped at the hello deadline — and is restarted on death. On
|
||||||
|
driver loss the compositor keeps the last frame (its `.scanout` calls now return
|
||||||
|
`-EPEER` instead of hanging — a kernel fix: an endpoint is marked dead when its owner
|
||||||
|
dies) and **re-attaches** when the restarted driver re-announces. A permanent give-up
|
||||||
|
(crash-loop cap) leaves the frozen frame; GOP is not re-taken.
|
||||||
|
- [x] `test/qemu_test.py`: the `virtio-gpu`, `display-native` (hot-attach), `display-modeset`,
|
||||||
|
and `display-reattach` (driver-kill/re-attach) cases. display-v2.md status updated.
|
||||||
|
|
||||||
|
**Gate (met):** the `display-reattach` case — device-manager (in `test-scanout-restart` mode)
|
||||||
|
kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||||
|
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||||
|
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||||
|
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||||
|
`supervision`, `shm`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||||
|
`display-modeset`) pass; default `zig build` is clean.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Deferred (explicitly not in this plan)
|
||||||
|
|
||||||
|
- **Client-rendered surfaces** — now unblocked by the `shm` capability (V2): an app renders
|
||||||
|
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||||
|
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||||
|
slots behind the same interface if wanted.
|
||||||
|
- **Real-GPU (NVIDIA/AMD/Intel) drivers** — out of scope; those devices stay on the GOP
|
||||||
|
floor by design.
|
||||||
|
- **Hardware-accelerated compositing / multiple heads** — future.
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
# The display service v2: a pluggable scanout backend
|
||||||
|
|
||||||
|
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||||
|
virtio-gpu driver announces itself, hot-attaches a native backend over the shared `shm`
|
||||||
|
scanout surface — with runtime mode-setting, EDID, and fenced (vsync) presents, and it
|
||||||
|
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||||
|
|
||||||
|
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||||
|
composites a layer stack into a cacheable back buffer and streams damage to the linear
|
||||||
|
framebuffer the firmware handed over. That path is portable and good: it drives any GPU,
|
||||||
|
including a real NVIDIA card at an ultrawide's native resolution, with zero GPU-specific
|
||||||
|
code. v2 keeps it as the **floor** and makes *scanout* — how a finished frame reaches the
|
||||||
|
panel — a **pluggable backend**, so the compositor can **upgrade to a real GPU driver when
|
||||||
|
one is present** and fall back to the framebuffer when it isn't.
|
||||||
|
|
||||||
|
The compositor itself (layers, back buffer, damage) does not change. Only the last step —
|
||||||
|
"put this frame on screen" — becomes swappable.
|
||||||
|
|
||||||
|
## The shape
|
||||||
|
|
||||||
|
```
|
||||||
|
compositor (display service) ── layer stack + back buffer + damage (unchanged)
|
||||||
|
│ composites a frame, then: backend.present(damage)
|
||||||
|
▼
|
||||||
|
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||||
|
│
|
||||||
|
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||||
|
│ Always available. No mode-set, no vsync. THE FLOOR.
|
||||||
|
│
|
||||||
|
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||||
|
service: present via a shared resource + flush (real vsync),
|
||||||
|
EDID mode list, runtime mode-set.
|
||||||
|
```
|
||||||
|
|
||||||
|
A **backend** is a small interface the compositor calls:
|
||||||
|
|
||||||
|
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||||
|
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||||
|
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||||
|
a virtio flush, optionally vsync-fenced, for the native path),
|
||||||
|
- capability queries — `canModeSet`, `hasVsync` — and, when supported, `modes()` /
|
||||||
|
`setMode(m)`.
|
||||||
|
|
||||||
|
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||||
|
today; everything device-specific lives behind the interface.
|
||||||
|
|
||||||
|
## Selection and hot-attach
|
||||||
|
|
||||||
|
The choice is **dynamic**, because a GPU driver is spawned asynchronously (the device
|
||||||
|
manager brings it up after boot), and because danos is meant to be resilient:
|
||||||
|
|
||||||
|
1. **Boot on GOP.** The compositor starts on `GopBackend` immediately, so there is never a
|
||||||
|
blank screen while drivers load — the exact v1 behaviour.
|
||||||
|
2. **Upgrade on announce.** When the virtio-gpu driver has claimed its device and set up a
|
||||||
|
scanout, it **announces itself to the display service** (a `push`: the driver looks up
|
||||||
|
`.display` and sends an *attach-scanout* message carrying its `scanout` endpoint as a
|
||||||
|
capability). The compositor switches to `VirtioGpuBackend` and re-presents the current
|
||||||
|
frame full-screen. Push beats polling — the compositor doesn't know a priori which
|
||||||
|
driver, if any, exists, and danos has no service-registration pub/sub.
|
||||||
|
3. **Native is restartable, not fallback-on-crash.** Once a native driver has reprogrammed
|
||||||
|
the device, the firmware's GOP framebuffer is **stale** — "native → GOP" is not a clean
|
||||||
|
fall-back. So a native driver that **crashes** is *restarted* by its supervisor (the
|
||||||
|
resilience work already merged), re-announces, and the compositor **re-attaches**
|
||||||
|
(native → native). The screen freezes on the last frame during the gap — acceptable.
|
||||||
|
4. **GOP is the floor for "no driver was ever there."** On a real GPU (NVIDIA/AMD/Intel)
|
||||||
|
the class-0x03 device matches nothing in the driver table, no `scanout` is ever
|
||||||
|
announced, and the compositor stays on GOP forever — no special-casing. Only if a
|
||||||
|
native driver *permanently* gives up (crash-loop cap) does the compositor attempt GOP
|
||||||
|
again, and even then only if the LFB is still mappable.
|
||||||
|
|
||||||
|
## The shared-memory primitive this needs
|
||||||
|
|
||||||
|
virtio-gpu's scanout resource is **guest RAM** — the driver allocates it and attaches it
|
||||||
|
to a virtio resource, and the compositor composes into it. That means the compositor
|
||||||
|
writing into the driver's buffer is **cross-process memory sharing**, the primitive v1
|
||||||
|
deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural generalization
|
||||||
|
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||||
|
|
||||||
|
```
|
||||||
|
shm_create(len) -> {handle, vaddr} // a shareable, page-aligned RAM region
|
||||||
|
… pass `handle` as the send_cap on an ipc_call …
|
||||||
|
shm_map(cap) -> vaddr // the receiver maps the same physical pages
|
||||||
|
```
|
||||||
|
|
||||||
|
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||||
|
client-rendered surfaces (an app composing its own bitmap and handing the compositor a
|
||||||
|
reference instead of drawing by command). One piece of kernel work, two features.
|
||||||
|
|
||||||
|
## The virtio-gpu driver
|
||||||
|
|
||||||
|
A new ring-3 driver process (the topology v1 anticipated — "split the driver from the
|
||||||
|
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||||
|
|
||||||
|
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||||
|
- creates a **2D scanout resource** backed by an `shm` region, `attach_backing`s it,
|
||||||
|
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||||
|
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||||
|
at a chosen mode for **runtime mode-setting**,
|
||||||
|
- registers a `scanout` service and announces to the display service.
|
||||||
|
|
||||||
|
Its `resource_flush` is the real **present** — and gives a genuine **vsync/tear-free**
|
||||||
|
path a dumb GOP framebuffer can't.
|
||||||
|
|
||||||
|
## What v2 unlocks — and its honest scope
|
||||||
|
|
||||||
|
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||||
|
refresh / bpp), **EDID** enumeration, and **vsync**. But only on devices we have a driver
|
||||||
|
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||||
|
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||||
|
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||||
|
the **pluggable architecture** (a driver slots in when one exists) and a **rich, vsync'd
|
||||||
|
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||||
|
|
||||||
|
## Locked decisions
|
||||||
|
|
||||||
|
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||||
|
present/flush (and vsync), and exercises the whole pluggable design. Tested with QEMU
|
||||||
|
`-device virtio-gpu`.
|
||||||
|
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||||
|
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||||
|
fall-back after a reprogram.
|
||||||
|
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||||
|
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||||
|
client-surface path.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||||
|
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||||
|
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||||
|
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||||
+264
@@ -0,0 +1,264 @@
|
|||||||
|
# The display service: a framebuffer compositor
|
||||||
|
|
||||||
|
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||||
|
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||||
|
into it directly. That console is a stop-gap. The **display service**
|
||||||
|
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||||
|
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||||
|
**presents** finished frames to the screen — the display half of the GUI track
|
||||||
|
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||||
|
|
||||||
|
This note is the architecture and the reasoning behind it. The concrete build order
|
||||||
|
lives in [display-plan.md](display-plan.md).
|
||||||
|
|
||||||
|
## First, a distinction that shapes everything: GOP vs. the PCI device
|
||||||
|
|
||||||
|
It is tempting to think "the GOP framebuffer" and "the VGA-compatible display
|
||||||
|
controller in the PCIe tree" are two different things. They are not — they are **two
|
||||||
|
interfaces to the same silicon, at different times and different levels**, and knowing
|
||||||
|
which one you're holding decides what you can do.
|
||||||
|
|
||||||
|
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||||
|
linear framebuffer pointer and can set video modes — but only until
|
||||||
|
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||||
|
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||||
|
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||||
|
mode list, no EDID. What survives is the frozen snapshot in
|
||||||
|
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||||
|
pitch, format}`, and nothing more.
|
||||||
|
|
||||||
|
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||||
|
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||||
|
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||||
|
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||||
|
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||||
|
VRAM BAR. danos already decodes this device
|
||||||
|
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||||
|
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||||
|
triple) — but nothing binds it yet.
|
||||||
|
|
||||||
|
What that difference costs you, concretely:
|
||||||
|
|
||||||
|
| You want to… | Dumb GOP framebuffer (boot handoff) | Native device driver (PCI 0x03) |
|
||||||
|
|-------------------------------------------|-------------------------------------|------------------------------------------|
|
||||||
|
| **Report** the current mode | ✅ from the handoff | ✅ |
|
||||||
|
| **Change resolution / bpp at runtime** | ❌ GOP is gone | ✅ program DISPI regs / virtio-gpu queue |
|
||||||
|
| **Re-read EDID, enumerate monitor modes** | ❌ | ✅ the device exposes an EDID block |
|
||||||
|
| **Refresh rate** | ❌ (virtual anyway) | only a real KMS driver — far future |
|
||||||
|
| **vblank / tear-free present** | ❌ no vblank signal | ✅ vblank IRQ + page-flip (real GPUs) |
|
||||||
|
| **Works on the Pi (no PCI VGA)** | ✅ VideoCore hands a simple FB | ✗ per-device |
|
||||||
|
|
||||||
|
The lesson: the **portable base for the whole GUI stack is the GOP / boot-handoff linear
|
||||||
|
framebuffer**. Runtime mode-setting is a *per-device upgrade* layered on top — and on
|
||||||
|
the Raspberry Pis there is no PCI VGA at all, so the neutral framebuffer is the only
|
||||||
|
thing all three target machines share. That is why the display service is built on the
|
||||||
|
dumb framebuffer first, with the native backend as an optional module behind the same
|
||||||
|
interface.
|
||||||
|
|
||||||
|
## Two constraints this service exists to meet
|
||||||
|
|
||||||
|
Like the input service — which existed partly to motivate the asynchronous
|
||||||
|
[`ipc_send`](ipc.md) primitive — the display service runs straight into two limits the
|
||||||
|
rest of the system hasn't had to face:
|
||||||
|
|
||||||
|
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||||
|
mapped into the kernel's physmap, and is touched only by
|
||||||
|
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||||
|
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||||
|
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||||
|
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||||
|
touch the pixels**. (See "The handoff" below — this is built.)
|
||||||
|
|
||||||
|
2. **danos has no cross-process shared memory.** The memory syscalls are `mmap`
|
||||||
|
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||||
|
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||||
|
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||||
|
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||||
|
use it — it would have to *map* another process's memory, which nothing allows. This
|
||||||
|
is deferred (see "What v1 does not do"), because v1 sidesteps it entirely.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ── owns the boot framebuffer; bootstrap console only
|
||||||
|
│ seeds a "display0" device node from BootInformation.framebuffer
|
||||||
|
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||||
|
│ plus DisplayInfo{width, height, pitch, format})
|
||||||
|
▼
|
||||||
|
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||||
|
│ device.claim(display0) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||||
|
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||||
|
│ owns: an ordered LAYER STACK + a per-frame DAMAGE list
|
||||||
|
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||||
|
│ backend is an INTERNAL interface: {gop-fb} today; {bochs-dispi, virtio-gpu} later
|
||||||
|
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||||
|
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||||
|
drawing clients (v1) surface clients (deferred)
|
||||||
|
runtime.display commands: runtime.display surfaces:
|
||||||
|
create_layer / configure_layer shm_create → pass as a capability →
|
||||||
|
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||||
|
present client-rendered bitmap directly
|
||||||
|
```
|
||||||
|
|
||||||
|
The bring-up sequence mirrors a hardware driver's — it is the
|
||||||
|
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||||
|
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||||
|
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||||
|
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||||
|
`extern struct` messages and an `Operation` tag).
|
||||||
|
|
||||||
|
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||||
|
composites — it does not split a "framebuffer driver" from a "compositor" the way input
|
||||||
|
splits `ps2-bus` from the input service. The backend (dumb FB vs. a native GPU) is an
|
||||||
|
*internal* interface, not a process boundary. That boundary earns its keep only when a
|
||||||
|
second backend or a second monitor appears; until then it is complexity with no payoff.
|
||||||
|
|
||||||
|
## The handoff: a device node + a write-combining map
|
||||||
|
|
||||||
|
The framebuffer crosses into user space through the machinery that already exists for
|
||||||
|
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||||
|
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||||
|
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||||
|
re-claims it).
|
||||||
|
|
||||||
|
- The kernel seeds a synthetic **`display0`** node into the
|
||||||
|
[devices-broker](../system/kernel/devices-broker.zig) at init, from
|
||||||
|
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||||
|
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||||
|
`DisplayInfo{width, height, pitch, format}` (the memory resource says *where* and *how
|
||||||
|
big*; `DisplayInfo` says how to *interpret* the bytes).
|
||||||
|
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||||
|
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||||
|
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||||
|
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||||
|
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||||
|
back→front blit unusably slow.
|
||||||
|
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||||
|
LFB. A panic is the one exception — by then the service is likely dead anyway, and a
|
||||||
|
panic on screen wins.
|
||||||
|
|
||||||
|
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||||
|
`vfs`/`input`/`device-manager` ([init.zig](../system/services/init/init.zig)), and it
|
||||||
|
self-discovers `display0` with `device.enumerate`. The [device manager](device-manager.md)
|
||||||
|
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||||
|
this singleton synthetic node.
|
||||||
|
|
||||||
|
## Double buffering and the write-combining discipline
|
||||||
|
|
||||||
|
Two buffers, with deliberately different memory types:
|
||||||
|
|
||||||
|
- The **front buffer** is the LFB — **write-combining**: fast to *write*, slow to
|
||||||
|
*read*. The rule is therefore **never read the front buffer**. Only ever stream into
|
||||||
|
it, sequentially.
|
||||||
|
- The **back buffer** is ordinary **cacheable** RAM (`mmap`), the same geometry. All
|
||||||
|
compositing happens here, where reads and read-modify-write blends are cheap.
|
||||||
|
|
||||||
|
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||||
|
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||||
|
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||||
|
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||||
|
|
||||||
|
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||||
|
|
||||||
|
These are two different artifacts, and the dumb framebuffer fixes exactly one of them:
|
||||||
|
|
||||||
|
- **Flicker** is the user seeing intermediate, half-drawn states (a clear-then-redraw
|
||||||
|
flash). Double buffering **eliminates it completely** — the screen only ever receives
|
||||||
|
whole, finished frames.
|
||||||
|
- **Tearing** is a present landing while the display's scanout beam is mid-frame, so the
|
||||||
|
top of the screen shows the new frame and the bottom the old. Avoiding it requires
|
||||||
|
presenting during the vertical blank (**vsync**) — which needs a vblank signal. **A
|
||||||
|
dumb GOP framebuffer has no vblank.**
|
||||||
|
|
||||||
|
So v1 is **flicker-free**, and it *minimizes* the tear window by presenting only damaged
|
||||||
|
rectangles (less to copy → a smaller window in which the beam can catch a half-updated
|
||||||
|
frame), but it is **not tear-free**. Genuine vsync waits for a backend with a vblank IRQ
|
||||||
|
or a flush/flip path — a native-device capability, not something the firmware
|
||||||
|
framebuffer can offer. Stated plainly here so the limitation is understood, not
|
||||||
|
discovered.
|
||||||
|
|
||||||
|
## Layers and the client protocol
|
||||||
|
|
||||||
|
The compositor holds an **ordered stack of layers**. Each layer has a rectangle, a
|
||||||
|
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||||
|
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||||
|
|
||||||
|
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||||
|
immediate-mode command protocol — essentially the model early X used, and enough for a
|
||||||
|
shell, a terminal, a cursor, and a wallpaper:
|
||||||
|
|
||||||
|
| Operation | Meaning |
|
||||||
|
|--------------------|---------------------------------------------------------------|
|
||||||
|
| `info` | report `{width, height, pitch, format}` of the display |
|
||||||
|
| `create_layer` | allocate a server-owned surface, return a layer handle |
|
||||||
|
| `configure_layer` | set a layer's rect, z-order, visibility |
|
||||||
|
| `destroy_layer` | release a layer |
|
||||||
|
| `fill_rect` | fill a rectangle of a layer with a colour |
|
||||||
|
| `blit_tile` | copy a small client-supplied pixel tile into a layer (inline) |
|
||||||
|
| `damage` | mark a region of a layer dirty |
|
||||||
|
| `present` | composite dirty layers and flush to the screen |
|
||||||
|
|
||||||
|
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||||
|
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||||
|
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||||
|
policy in the client.
|
||||||
|
|
||||||
|
## `runtime.display`
|
||||||
|
|
||||||
|
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||||
|
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||||
|
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||||
|
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||||
|
runtime, as with every other danos service.
|
||||||
|
|
||||||
|
## What v1 does not do (and why that's fine)
|
||||||
|
|
||||||
|
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
||||||
|
both are clean additions behind the interfaces v1 establishes.
|
||||||
|
|
||||||
|
- **Client-rendered surfaces (shared memory).** The fast path for a bitmap-heavy app is
|
||||||
|
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||||
|
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||||
|
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||||
|
from *endpoints* to *memory objects* (`shm_create(len) → {cap, vaddr}`, pass `cap` on
|
||||||
|
an `ipc_call`, receiver `shm_map(cap) → vaddr`). v1 avoids it because server-owned
|
||||||
|
surfaces already prove the whole pipeline.
|
||||||
|
|
||||||
|
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||||
|
resolution / bpp at runtime needs the raw PCI device. The first native backend is
|
||||||
|
Bochs DISPI — the register interface QEMU's `-device VGA` exposes — behind the same
|
||||||
|
internal backend interface the dumb framebuffer sits behind. Refresh-rate and colour
|
||||||
|
management (a gamma LUT) are real-GPU-KMS territory, far beyond this.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
Three QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||||
|
test/qemu_test.py <case>`), each layering on the last:
|
||||||
|
|
||||||
|
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||||
|
the claim → `mmio_map` leaf is genuinely **write-combining** (PAT entry 4), asserted at
|
||||||
|
the page-table level.
|
||||||
|
- **`display-service`** — the compositor comes up: it claims the framebuffer, allocates
|
||||||
|
the cacheable back buffer, presents a cleared frame through the double-buffer path
|
||||||
|
(`display: online … / presented frame 0`), and a startup **self-check** composites two
|
||||||
|
overlapping layers on the real framebuffer and reads them back — overlap = the top
|
||||||
|
layer — logging `display: compositor self-check ok`.
|
||||||
|
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||||
|
[`display-demo`](../system/services/display-demo/) client (the
|
||||||
|
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper, a
|
||||||
|
sliding rectangle, a cursor — through the layer client API and heartbeats
|
||||||
|
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||||
|
the [input test](input.md) proves an event travels source → service → subscriber. The
|
||||||
|
visible motion itself is a screenshot away via `zig build run-x86-64`.
|
||||||
|
|
||||||
|
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||||
|
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||||
|
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||||
|
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||||
|
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||||
|
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||||
|
- [display-plan.md](display-plan.md) — the ordered build-out.
|
||||||
+10
-9
@@ -57,8 +57,9 @@ is not an address window. Discovery is trusted; user space is not.
|
|||||||
|
|
||||||
### What a bus driver looks like
|
### What a bus driver looks like
|
||||||
|
|
||||||
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and
|
||||||
its "devices" are the block's comparators:
|
`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block
|
||||||
|
as the "bus" and its comparators as the "devices", is:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
_ = dev.claim(bus.id); // 1. own the bus
|
_ = dev.claim(bus.id); // 1. own the bus
|
||||||
@@ -78,8 +79,8 @@ for (0..n) |i| { // 3. publish each child
|
|||||||
|
|
||||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||||
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
whose window escapes the bus is refused; the in-kernel `containment` test asserts the
|
||||||
asserts the kernel's table upholds it.
|
kernel's table upholds that ([drivers.md](drivers.md)).
|
||||||
|
|
||||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||||
through its controller, not by MMIO. That case is allowed and is the common one.
|
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||||
@@ -143,7 +144,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
|||||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||||
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||||
no DMA driver consumes `dma_alloc` yet.
|
no DMA driver consumes `dma_alloc` yet.
|
||||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||||
@@ -302,8 +303,8 @@ rather than an out-struct. The rest of this section is the original design note.
|
|||||||
|
|
||||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||||
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||||
device has.
|
device has.
|
||||||
|
|
||||||
**The fix, in two halves.**
|
**The fix, in two halves.**
|
||||||
@@ -326,7 +327,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i
|
|||||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||||
|
|
||||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||||
exercise this path. The first MSI driver will be the first PCI driver.
|
exercise this path. The first MSI driver will be the first PCI driver.
|
||||||
|
|
||||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||||
@@ -353,7 +354,7 @@ gap should be named rather than implied.
|
|||||||
|
|
||||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||||
could write a real one against `bus`'s comparators tomorrow.
|
could write a real one against any device a bus driver publishes tomorrow.
|
||||||
|
|
||||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||||
|
|||||||
+40
-29
@@ -22,12 +22,12 @@ say.*
|
|||||||
|
|
||||||
## How a driver gets started: discover, match, spawn
|
## How a driver gets started: discover, match, spawn
|
||||||
|
|
||||||
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that
|
||||||
and policy lives in user space. Boot brings user space up as a three-level supervision
|
is policy, and policy lives in user space. Boot brings user space up as a three-level
|
||||||
hierarchy, each level owning one job:
|
supervision hierarchy, each level owning one job:
|
||||||
|
|
||||||
```
|
```
|
||||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus
|
||||||
| | |
|
| | |
|
||||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||||
publishes the starts the system /system/devices, matches each device
|
publishes the starts the system /system/devices, matches each device
|
||||||
@@ -184,7 +184,11 @@ Two properties worth knowing:
|
|||||||
|
|
||||||
## A whole driver
|
## A whole driver
|
||||||
|
|
||||||
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such
|
||||||
|
example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`,
|
||||||
|
`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a
|
||||||
|
compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the
|
||||||
|
shape is:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||||
@@ -209,8 +213,8 @@ while (...) {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a
|
||||||
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
clocksource — the only way to use it is to read it, so it exercises `mmio_map` without
|
||||||
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||||
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||||
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||||
@@ -252,9 +256,10 @@ bus driver may only ever subdivide what it already owns.
|
|||||||
A device with **no resources** is legal and common. A USB device is reached through its
|
A device with **no resources** is legal and common. A USB device is reached through its
|
||||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||||
|
|
||||||
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||||
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||||
controller drivers fit together.
|
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||||
|
drivers and host controller drivers fit together.
|
||||||
|
|
||||||
## What the kernel does not do for you
|
## What the kernel does not do for you
|
||||||
|
|
||||||
@@ -313,32 +318,38 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
|||||||
|
|
||||||
## Verifying it
|
## Verifying it
|
||||||
|
|
||||||
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
No demo driver ships to prove this end to end; the *real* drivers do, so the tests
|
||||||
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
target them and the kernel primitives directly:
|
||||||
through `replyWait` returning a notification — it cannot reach that line by polling.
|
|
||||||
|
|
||||||
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
- **`device-manager`** — boots only the device manager, which discovers the PCI host
|
||||||
APIC redirection entry back and asserts the line really is routed to a device vector,
|
bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the
|
||||||
really is level-triggered, and really was left unmasked by the driver's final
|
process table and the device tree — to confirm pci-bus came up and registered the
|
||||||
`irq_ack`.
|
functions it enumerated: the whole discover → match → spawn → driver-up chain.
|
||||||
|
- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ,
|
||||||
|
delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end.
|
||||||
|
- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM
|
||||||
|
window) and walks it: `mmio_map`, end to end.
|
||||||
|
- **`containment`** — the kernel refuses a `device_register` whose child window escapes
|
||||||
|
the parent's grant (else it would be a syscall for mapping arbitrary memory), while an
|
||||||
|
identical re-register stays idempotent. Asserted in-kernel, straight against the broker.
|
||||||
|
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||||
|
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||||
|
is not. That second half is why bindings are keyed on the owning *task* and not on the
|
||||||
|
endpoint pointer — endpoints are shared, so releasing "everything pointing at this
|
||||||
|
endpoint" would silently mask a live driver's device.
|
||||||
|
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||||
|
space never returns MMIO frames to the RAM pool.
|
||||||
|
|
||||||
```
|
```
|
||||||
$ python3 test/qemu_test.py hpet irqfree iopass
|
$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass
|
||||||
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
acpi-ps2 ... PASS
|
||||||
|
pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
containment ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
```
|
```
|
||||||
|
|
||||||
Two companions cover what `hpet` can't, because it never exits:
|
|
||||||
|
|
||||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
|
||||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
|
||||||
is not. That second half is why bindings are keyed on the owning *task* and not on
|
|
||||||
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
|
||||||
this endpoint" would silently mask a live driver's device.
|
|
||||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
|
||||||
space never returns MMIO frames to the RAM pool.
|
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||||
|
|||||||
@@ -0,0 +1,556 @@
|
|||||||
|
# Native Intel iGPU display support — feasibility and roadmap
|
||||||
|
|
||||||
|
**Status: research snapshot, not implemented.** This records what a *minimal, display-only*
|
||||||
|
native driver for an **Intel integrated GPU** — EDID read + mode-set + framebuffer scanout, with
|
||||||
|
**no** 3D/media/compute — would take, and how it slots into danos's pluggable scanout
|
||||||
|
architecture. It is a survey of primary sources (Intel's open-source
|
||||||
|
[Programmer's Reference Manuals](https://www.intel.com/content/www/us/en/docs/graphics-for-linux/developer-reference/1-0/overview.html),
|
||||||
|
coreboot's [libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html), the Linux
|
||||||
|
[i915 display](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/i915/display) driver,
|
||||||
|
and Haiku's [intel_extreme](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)),
|
||||||
|
not an implementation. It is the companion to [nvidia-gpus.md](nvidia-gpus.md) and should be read
|
||||||
|
against it — the two answer the same question for opposite silicon.
|
||||||
|
|
||||||
|
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the v2
|
||||||
|
model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||||
|
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||||
|
|
||||||
|
## TL;DR
|
||||||
|
|
||||||
|
- **Intel is a materially easier, lower-tier target than the NVIDIA RTX 3060 — and the reason is
|
||||||
|
documentation, not silicon.** Intel publishes official, register-level, per-platform **Display
|
||||||
|
Engine** PRMs with named registers, bitfields, and numbered enable sequences; NVIDIA publishes
|
||||||
|
no display PRM and forces reverse-engineering against GPL nouveau. A minimal Intel display-only
|
||||||
|
driver is roughly **tier 2 to low-tier 3** for well-covered generations (Skylake / Kaby Lake /
|
||||||
|
Coffee Lake), versus NVIDIA's **tier 4** for GA106. This is the load-bearing conclusion.
|
||||||
|
- **The display block is a genuinely separable register domain.** Mode-set + scanout touch only
|
||||||
|
display registers (pipes, planes, transcoders, DDI buffers, PLLs, power wells, GMBUS/AUX) — **no
|
||||||
|
render engine, no command streamer, no GEM/3D, no signed microcode.** Two small carve-outs, both
|
||||||
|
trivial pokes that do *not* pull in the render engine: a real CDCLK frequency change writes the
|
||||||
|
shared GT PCODE mailbox, and the plane's surface register is a GGTT (memory-interface) address.
|
||||||
|
- **There is no firmware wall on the display path.** The only display microcontroller (DMC / "CSR",
|
||||||
|
Skylake+) is **optional** — its sole job is saving/restoring display state across DC5/DC6
|
||||||
|
low-power idle. Without it, i915 prints "Disabling runtime power management" and mode-sets and
|
||||||
|
scans out normally. GuC/HuC are render/media coprocessors, never touched by a display driver.
|
||||||
|
Pre-Skylake parts have no display microcontroller at all yet mode-set fine. There is **nothing
|
||||||
|
analogous to NVIDIA's GSP**.
|
||||||
|
- **The scanout memory model is dramatically simpler than a discrete GPU.** Intel iGPUs have **no
|
||||||
|
VRAM**: the display scans out of ordinary system RAM addressed through the Global GTT (GGTT), a
|
||||||
|
flat single-level page table. Linear (untiled) framebuffers are first-class. You need **no
|
||||||
|
GEM/TTM, no VMM, no VRAM allocator, no BAR1 aperture juggling** — the exact machinery the NVIDIA
|
||||||
|
path forces on you.
|
||||||
|
- **coreboot libgfxinit is a compact, complete, display-only reference** doing precisely this scope
|
||||||
|
(EDID + PLL/mode-set + scanout, zero 3D) in ~22k lines of formally-analysed SPARK/Ada — versus
|
||||||
|
i915's ~400k lines. It is a *read-and-reimplement* reference, not drop-in code (GPL-2.0-or-later,
|
||||||
|
and Ada, not Zig).
|
||||||
|
- **The clean-room, permissively-licensed path is real** — you can implement from the PRM without
|
||||||
|
reading GPL code, and Haiku's MIT `intel_extreme` is a permissive precedent. This is the decisive
|
||||||
|
contrast with NVIDIA, where no vendor register spec exists.
|
||||||
|
- **The practical catch is hardware, not software.** On a desktop with an RTX 3060, the monitor is
|
||||||
|
almost certainly cabled to the *card*, so an iGPU driver would light a dark motherboard port; the
|
||||||
|
CPU may be an **F-SKU with the iGPU fused off entirely**; and every clean-room reference targets
|
||||||
|
*older* Intel. Intel is the right target to **learn** display bring-up — "run it on my machine"
|
||||||
|
is a separate, machine-dependent question that may not resolve in the reader's favour.
|
||||||
|
- **Recommendation:** as with the NVIDIA doc, GOP already gives native-resolution scanout with zero
|
||||||
|
GPU code. A native Intel driver buys runtime mode changes, hardware vsync, and multihead — and it
|
||||||
|
reaches "first pixel" far faster than the NVIDIA path *if* the target machine actually has a
|
||||||
|
usable, cable-attached iGPU of a documented generation.
|
||||||
|
|
||||||
|
## Display engine architecture, and why it's separable
|
||||||
|
|
||||||
|
For the common single-display path (SST DisplayPort / HDMI / eDP), the Intel display data flow is a
|
||||||
|
small, fully documented, essentially fixed sequence:
|
||||||
|
|
||||||
|
```
|
||||||
|
memory surface → PLANE(s) → PIPE → TRANSCODER → DDI (drives IO/PHY) → connector
|
||||||
|
```
|
||||||
|
|
||||||
|
The Tiger Lake PRM Vol 12 states it verbatim: *"The front end of the display contains the pipes.
|
||||||
|
The pipes connect to the transcoders. The transcoders, except for wireless, connect to the DDIs to
|
||||||
|
drive the IO/PHY."* A **pipe** blends planes (primary/sprite/cursor) into one raster stream; the
|
||||||
|
**transcoder** wraps it in port-protocol timing (DP/HDMI/eDP/DSI); the **DDI** is the physical port
|
||||||
|
and PHY. Pipe, Planes, Transcoder, and Digital Display Interface are each first-class PRM chapters
|
||||||
|
with per-object files in libgfxinit
|
||||||
|
([TGL PRM Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||||
|
|
||||||
|
**Two honest qualifications** the raw research overstated (per verification):
|
||||||
|
|
||||||
|
- The pipeline is *not* strictly linear in all cases — the same PRM pages document optional branches
|
||||||
|
a minimal driver simply ignores (wireless writeback to memory, MIPI DSI, DisplayPort multistream
|
||||||
|
many-to-one, DSC/tiled pipe-joining). Ignoring them does not weaken feasibility.
|
||||||
|
- The four-object model *as named* is **Haswell-onward** (DDI introduced ~2013), not "every gen."
|
||||||
|
Pre-Haswell used FDI + PCH transcoders + port-specific encoders. Within the modern iGPU range
|
||||||
|
danos would realistically target (Skylake → Meteor/Lunar Lake) the model is stable.
|
||||||
|
|
||||||
|
**The DPLL/clock block is a separate, per-port programmable clock source** and is one of the harder,
|
||||||
|
most gen-specific pieces: pick/enable a PLL, route its output to the DDI, then bring up the port.
|
||||||
|
The register layout and divider math change substantially per generation — pre-SKL SPLL/WRPLL/LCPLL,
|
||||||
|
Skylake+ shared DPLL0–3, Gen11+ combo-PHY plus Type-C MG/DKL PLLs. Pixel-clock computation is a
|
||||||
|
classic per-gen rewrite.
|
||||||
|
|
||||||
|
### Separable from render — the single most important enabler
|
||||||
|
|
||||||
|
The display is a distinct register domain from render/media, and this is confirmed at the primary
|
||||||
|
level: the TGL PRM ships display as its own volume (Vol 12), separate from Render Engine (Vol 9) and
|
||||||
|
Media (Vol 11); Linux's KMS "is provided by Intel Display Driver, and **shared with drm/xe**"
|
||||||
|
([kernel.org i915](https://docs.kernel.org/gpu/i915.html)) — i.e. the display module is
|
||||||
|
reused across two different GPU drivers. A full mode-set lights a display end-to-end using only power
|
||||||
|
wells, PLL/port-clock, DDI-buffer/PHY, transcoder and pipe registers — **zero render commands, zero
|
||||||
|
GEM objects, zero command-streamer.** libgfxinit is decisive proof: complete EDID + modeset +
|
||||||
|
framebuffer with no render/3D code at all.
|
||||||
|
|
||||||
|
Two carve-outs the "touches ONLY display registers" phrasing needs (per verification), **neither of
|
||||||
|
which drags in the render engine**:
|
||||||
|
|
||||||
|
1. A mode-set that changes the **Core Display Clock (CDCLK)** frequency/voltage pokes the shared **GT
|
||||||
|
Driver Mailbox** (PCODE/PCU power-controller interface), per Vol 12's own "Display Voltage
|
||||||
|
Frequency Switching" step. A trivial register handshake, documented alongside the display sequence.
|
||||||
|
2. The primary plane's surface register (`PLANE_SURF`) holds a **GGTT graphics address** (a
|
||||||
|
memory-interface concept, not covered in Vol 12). Using pre-mapped stolen memory — as libgfxinit
|
||||||
|
does — sidesteps any active GGTT programming. See [Memory and scanout](#memory-and-scanout).
|
||||||
|
|
||||||
|
### Per-gen churn: what's stable, what you rewrite
|
||||||
|
|
||||||
|
The **object model** (pipes/planes/transcoders/DDIs, GMBUS-for-EDID, double-buffered plane registers
|
||||||
|
armed atomically) is conceptually stable from Ironlake/Haswell through Tiger Lake. What you rewrite
|
||||||
|
per generation is:
|
||||||
|
|
||||||
|
1. the **CPU-vs-PCH split and interconnect**,
|
||||||
|
2. the **port/PHY + DPLL** programming,
|
||||||
|
3. **register offsets + power-well / CDCLK topology**, and
|
||||||
|
4. the **mode-set enable sequence itself** (power-well ordering, PLL lock, DDI-buffer enable,
|
||||||
|
transcoder clock-select) — an effective fourth axis the raw research folded into (1)/(2).
|
||||||
|
|
||||||
|
Interconnect eras, with the timeline **corrected** (the cited Haiku doc was chronologically loose):
|
||||||
|
|
||||||
|
- **Gen5 Ironlake (2010) → Ivy Bridge:** FDI (Flexible Display Interface) links the CPU display
|
||||||
|
engine to PCH-resident ports. The FDI/PCH-split era begins at **Ironlake**, not Gen7.
|
||||||
|
- **Haswell (Gen7.5):** the main digital outputs come **back onto the CPU die as DDIs** (DDI A = eDP)
|
||||||
|
— the *opposite* of "moving output to the PCH," and it collapses the FDI/PCH dance **for the
|
||||||
|
digital ports only**. FDI is **retained** for the legacy VGA/CRT path (DDI E → PCH CRT DAC), so a
|
||||||
|
driver gets the single DDI code path only by omitting analog VGA (which a minimal driver does).
|
||||||
|
- **Skylake (Gen9):** reworks clock/PLL, CDCLK, and the power-well model; introduces the optional DMC.
|
||||||
|
- **Gen11 Ice Lake / Gen12 Tiger Lake:** add combo-PHY + USB-Type-C/Thunderbolt MG/DKL PHYs — the
|
||||||
|
single biggest cost increase, and the reason "newest silicon" is *not* the easiest target. (DSC is
|
||||||
|
documented per-**pipe**; MSO is an eDP feature — not "per-transcoder" as the raw research said.)
|
||||||
|
|
||||||
|
### The tractable sweet spot
|
||||||
|
|
||||||
|
The documented, tractable sweet spot for a from-scratch display-only driver is the
|
||||||
|
**Haswell (Gen7.5) / Broadwell (Gen8) DDI family, with Skylake (Gen9) as the modern-hardware pick**
|
||||||
|
since it shares the same DDI object model. Rationale:
|
||||||
|
|
||||||
|
- Broadwell has a complete, freely downloadable
|
||||||
|
[PRM Vol 11 Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf);
|
||||||
|
its engine (3 pipes A/B/C, 4 transcoders incl. transcoder-EDP that floats onto any pipe, DDI A–E,
|
||||||
|
WRPLL/SPLL/LCPLL) is the classic "DDI + transcoder + WRPLL" model.
|
||||||
|
- It predates the combo-PHY / Type-C / MG-DKL complexity of Ice Lake / Tiger Lake.
|
||||||
|
- libgfxinit's DDI **connector/EDID/DP layer is uniform from Haswell through Coffee Lake**, so the
|
||||||
|
hardest-to-get-right port logic generalises widely.
|
||||||
|
|
||||||
|
Two supporting claims from the raw research are **wrong and corrected here (verification):**
|
||||||
|
|
||||||
|
- **The BDW and SKL PRMs are NOT 0BSD-licensed.** Both carry a Creative Commons
|
||||||
|
**Attribution-NoDerivatives** notice. Only the *newer* OSRC PRMs (Tiger Lake 2021 onward) put their
|
||||||
|
embedded code samples under **Zero-Clause BSD**. So for the recommended Haswell/Broadwell/Skylake
|
||||||
|
generations there are no "copy-pasteable 0BSD code samples" — the legal basis is *reimplementation
|
||||||
|
from a CC-BY-ND spec* (register facts are not copyrightable), not copying.
|
||||||
|
- **FDI+PCH is not fully eliminated on Haswell/Broadwell.** The BDW PRM keeps FDI for the DDI E → PCH
|
||||||
|
CRT DAC. The "one DDI code path" holds only for the digital outputs a minimal driver targets.
|
||||||
|
|
||||||
|
Sandy/Ivy Bridge (Gen6/7) is where the hobby-doc walkthroughs concentrate (the OSDev GMBUS/EDID
|
||||||
|
material) but carries the FDI+PCH split cost. *(Low confidence on the OSDev specifics — the wiki
|
||||||
|
returns 403 to automated fetches and its "guaranteed to work" phrasing is a hobby assertion, not a
|
||||||
|
silicon guarantee.)*
|
||||||
|
|
||||||
|
## Documentation — and the clean-room question
|
||||||
|
|
||||||
|
This is the crux of the whole comparison. **Intel hands you the register spec that NVIDIA withholds.**
|
||||||
|
|
||||||
|
- The Tiger Lake **"Vol 12: Display Engine"** PRM is a real, first-party, open-source document —
|
||||||
|
**433 pages, verified by direct download** — with named registers + addresses + bitfield tables
|
||||||
|
(`TRANS_DDI_FUNC_CTL`, `DDI_BUF_CTL`, `DP_TP_CTL`, `PLANE_STRIDE`, `DPLL_CFGCR0/1`, `CDCLK_CTL`,
|
||||||
|
`PWR_WELL_CTL_DDI`, …) and **numbered, step-by-step enable sequences** with explicit writes, wait
|
||||||
|
conditions, and microsecond timeouts. It even includes the "magic value" tables older PRMs deferred
|
||||||
|
to the driver (DisplayPort PLL DCO/divider values; voltage-swing/de-emphasis in mV). *"A spec you
|
||||||
|
could write a driver from directly"* is well-supported, not hyperbole
|
||||||
|
([TGL Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||||
|
- **Clean-room, permissively-licensed implementation is legally and practically feasible from the
|
||||||
|
PRM alone.** CC-BY-ND governs redistribution of the *document*; register addresses and bit
|
||||||
|
definitions are functional facts, and original code implementing a described hardware interface is
|
||||||
|
not a derivative of the PDF. *(This is standard copyright reasoning, not adjudicated case law —
|
||||||
|
treat it as well-grounded, not settled.)* Two independent implementations already exist built
|
||||||
|
essentially from these docs (libgfxinit, Haiku), so the spec is demonstrably sufficient.
|
||||||
|
|
||||||
|
**The documentation ceiling — corrected.** The raw research said public PRMs stop "roughly at Ice
|
||||||
|
Lake / Tiger Lake." Verification refuted this: full public **"Vol 12 Display Engine"** PRMs exist for
|
||||||
|
Ice Lake, Lakefield, Tiger Lake, Rocket Lake, DG1, **and DG2/Arc "Alchemist" (Gen12.5, 2022)** —
|
||||||
|
[the ACM display PRM is public](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf).
|
||||||
|
The genuine cliff is **Meteor Lake (2023) and newer**: those have only a high-level architecture
|
||||||
|
overview, no register-level display PRM, and i915 references their display registers by opaque
|
||||||
|
internal **Bspec numeric IDs**. Alder Lake and Raptor Lake iGPUs are Gen12 Xe-LP display — the same
|
||||||
|
IP as Tiger Lake — so despite lacking a dedicated PRM they are effectively covered by the TGL PRM.
|
||||||
|
|
||||||
|
Net: a from-docs driver can confidently target **Skylake through DG2/Arc**, which is essentially the
|
||||||
|
entire current laptop/NUC installed base; only Meteor Lake and later slide back toward the NVIDIA
|
||||||
|
situation (reverse-engineering or reading GPL i915). The PRMs also survived 01.org's shutdown and are
|
||||||
|
mirrored in several stable places (Intel's cdrdv2 host, the
|
||||||
|
[Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) for Gen4–Gen9.5,
|
||||||
|
[kiwitree](https://kiwitree.net/~lina/intel-gfx-docs/prm/), x.org) — not a single point of failure.
|
||||||
|
*(Note: the Igalia archive stops at Kaby Lake and contains no Display Engine volume; the TGL/DG2
|
||||||
|
display PRMs are separate Intel/x.org downloads.)*
|
||||||
|
|
||||||
|
## coreboot libgfxinit — the native reference
|
||||||
|
|
||||||
|
[libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html) is the closest thing to a template danos
|
||||||
|
could ask for: a self-contained **native modeset library** (no VBIOS/int10, no firmware blobs) that
|
||||||
|
probes displays via EDID over DDC/I²C and DP AUX, and drives LVDS, eDP, DP1–3, HDMI1–3, analog VGA,
|
||||||
|
plus USB-C DP/HDMI alt-mode on Tiger Lake. It sets up pipes (Primary/Secondary/Tertiary), planes,
|
||||||
|
transcoders, PLLs, panel power/backlight, the GTT, and framebuffer scanout — **display-only, zero
|
||||||
|
3D/media/compute**, which is exactly danos's scope. Its public entry is essentially
|
||||||
|
`Initialize()` then `Update_Outputs(Pipe_Configs)`, where each `Pipe_Config` carries
|
||||||
|
`{Port, Framebuffer, Cursor, Mode}` — a near-perfect fit for a pluggable scanout backend.
|
||||||
|
|
||||||
|
Why it beats i915 as a reference (**verified by measurement**): **131 Ada source files, ~818 KB,
|
||||||
|
~22k code lines** across *all* generations, factored precisely along the axes you care about (`edid`,
|
||||||
|
`dp_aux`, `dp_training`, `pipe_setup`, `transcoder`, `plls`, `connectors`, `port_detect`), with
|
||||||
|
**none** of the DRM/KMS/GEM/TTM, GT/3D, RC6/RPS, or GuC/HuC machinery that makes
|
||||||
|
`drivers/gpu/drm/i915` **~419k lines / 900 files / 12 MB**. (A grep confirms *zero* gem/ttm/guc/huc/
|
||||||
|
execbuf identifiers in the tree.) It depends only on a small HW-access shim, `libhwbase`
|
||||||
|
(`HW.PCI`, `HW.Port_IO`, `HW.MMIO`, `HW.Time`), which maps naturally onto danos's MMIO-grant + IPC
|
||||||
|
primitives — you provide Zig equivalents and the modeset logic sits on top. *(Correction to the raw
|
||||||
|
research: the widely-quoted "~13–14k LOC" is only the generic `common/` layer; the eight
|
||||||
|
per-generation subdirs roughly double it.)*
|
||||||
|
|
||||||
|
**It is a read-and-reimplement reference, not drop-in code.** Two hard constraints:
|
||||||
|
|
||||||
|
- **License is GPL-2.0-or-later** (the COPYING file is GPLv2; per-file headers add "or any later
|
||||||
|
version"). The CC-BY-4.0 on the docs *site* is a footer, not the source license. Copyleft applies
|
||||||
|
to ported code.
|
||||||
|
- **It is SPARK/Ada, and designed to run as coreboot boot-firmware**, not a runtime OS driver. A
|
||||||
|
danos port means either an Ada/GNAT toolchain in the build or hand-transliteration into Zig; the
|
||||||
|
SPARK "absence of runtime errors" proof does **not** carry over to your reimplementation (and note
|
||||||
|
it proves absence of runtime errors, **not** functional modeset correctness).
|
||||||
|
|
||||||
|
Two more caveats worth knowing: its **error handling is limited** — "only the case that no display
|
||||||
|
could be found counts as failure"; a later DP link-training failure is *not* propagated. And its
|
||||||
|
**verified-in-coreboot** hardware list stops at **Coffee Lake + Apollo Lake**, even though the tree
|
||||||
|
contains a `tigerlake/` directory (Ice Lake has no directory at all, and Alder Lake support is only
|
||||||
|
"begun"). So treat Haswell..Coffee Lake as the trustworthy transliteration window and TGL as
|
||||||
|
present-but-less-proven.
|
||||||
|
|
||||||
|
The orchestration reads as a clean state machine (`hw-gfx-gma.adb` `Enable_Output`):
|
||||||
|
`Fill_Port_Config → Preferred_Link_Setting → PLLs.Alloc → [retry] Connectors.Pre_On →
|
||||||
|
Display_Controller.On → Connectors.Post_On`, with a literal *"try each DP-lane configuration twice"*
|
||||||
|
inner retry and an outer link-setting step-down. `hw-gfx-dp_training.adb` (398 lines) is a complete,
|
||||||
|
generic DP link-training implementation (TP1/TP2/TP3, CR + EQ loops, swing/pre-emphasis adjust from
|
||||||
|
sink status). Per-generation buffer translations plug in underneath via
|
||||||
|
`Program_Buffer_Translations`, gated on `Config.Has_DDI_Buffer_Trans`. All of this was confirmed
|
||||||
|
against the source line-by-line.
|
||||||
|
|
||||||
|
## The EDID + mode-set path (Haswell/Broadwell target)
|
||||||
|
|
||||||
|
The whole path is memory-mapped register programming with polled status bits — no command ring, no
|
||||||
|
microcode, no DMA channel.
|
||||||
|
|
||||||
|
**EDID over DDC (GMBUS).** Pure MMIO poking of the GMBUS I²C controller (`GMBUS0`–`GMBUS5`): `GMBUS0`
|
||||||
|
selects pin-pair/port + clock; `GMBUS1` carries slave address (`0x50` for EDID), byte count,
|
||||||
|
direction, SW-ready; `GMBUS2` exposes HW-ready/NAK/ACTIVE to poll; `GMBUS3` is a 4-byte data FIFO;
|
||||||
|
`GMBUS5` gives the 2-byte segment index for E-DDC. A read is: write `GMBUS0`, write `GMBUS1`
|
||||||
|
(`CYCLE_WAIT | count | SLAVE_READ | SW_RDY | slave<<addr`), loop {poll `HW_RDY`, read 4 bytes}, then
|
||||||
|
STOP ([i915 intel_gmbus.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_gmbus.c)).
|
||||||
|
|
||||||
|
**EDID + DPCD over DP AUX.** For DisplayPort/eDP, EDID (as I²C-over-AUX to `0x50`) and all DPCD
|
||||||
|
capability/link-status registers are read over the AUX channel: per-DDI `DDI_AUX_CTL` + 5×
|
||||||
|
`DDI_AUX_DATA`. Build a 3–5 byte header + payload, set SEND_BUSY, poll it clear, read
|
||||||
|
DONE/TIMEOUT/RECEIVE_ERROR. Message size 1–20 bytes; spec requires ≥3 retries. On Haswell/BDW the AUX
|
||||||
|
clock divider is programmed explicitly; SKL+ derive it automatically
|
||||||
|
([i915 intel_dp_aux.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dp_aux.c)).
|
||||||
|
Both GMBUS and DP-AUX live in libgfxinit's shared `common/` — cheap and nearly gen-invariant.
|
||||||
|
|
||||||
|
**The mode-set is a fixed, documented register sequence.** The Broadwell DisplayPort enable order
|
||||||
|
(verbatim from BDW PRM Vol 11, pp.98–99): (1) DDI lane capability; (2) panel power sequencing if
|
||||||
|
needed; (3) enable the CPU display PLL (WRPLL/SPLL) and wait ~20 µs; (4) Port Clock Select → DDI,
|
||||||
|
enable `DP_TP_CTL` with training pattern 1, configure `DDI_BUF_TRANS`, enable `DDI_BUF_CTL`, wait
|
||||||
|
>518 µs, run link training, set `DP_TP_CTL` to Normal (Idle first for eDP); (5) Transcoder Clock
|
||||||
|
Select, enable the plane, panel fitter if needed, program transcoder timings + M/N/TU, enable
|
||||||
|
`TRANS_DDI_FUNC_CTL`, enable `TRANS_CONF`, then backlight. Disable is the exact reverse — a bounded
|
||||||
|
checklist.
|
||||||
|
|
||||||
|
**DisplayPort/eDP link training is driver-driven in software over AUX** — the CPU runs the
|
||||||
|
clock-recovery and channel-equalization state machines by hand; it is **not** offloaded to a hardware
|
||||||
|
sequencer or firmware. The source side exposes only primitives: `DP_TP_CTL` selects the training
|
||||||
|
pattern the port emits; `DDI_BUF_CTL`/`DDI_BUF_TRANS` set voltage-swing/pre-emphasis. The driver
|
||||||
|
loops: emit pattern + set source levels → write `TRAINING_PATTERN_SET` (DPCD 0x102) + `TRAINING_LANEx_SET`
|
||||||
|
(0x103) over AUX → delay (100 µs CR / 400 µs EQ) → read `LANE_STATUS` → on failure adjust to the
|
||||||
|
sink's `ADJUST_REQUEST` values and retry. A few hundred lines of ordinary CPU/AUX code (libgfxinit
|
||||||
|
`Train_DP`: CR loop 1..32, EQ loop 1..6). **This is the single fiddliest, most fragile piece** — a
|
||||||
|
TMDS/HDMI panel avoids it entirely, and targeting an already-lit eDP panel avoids most of it.
|
||||||
|
|
||||||
|
**The clock (WRPLL) is documented divider math, not a magic table.** On Haswell/BDW the WRPLL derives
|
||||||
|
the symbol clock from a 2700 MHz LCPLL reference through R2/N2/P dividers with VCO 2400–4800 MHz —
|
||||||
|
small integer arithmetic. DP is *easier* than HDMI because it runs at a few fixed link rates (1.62 /
|
||||||
|
2.7 / 5.4 GHz), so a DP/eDP-only minimal driver can often use fixed rates and skip most of the search.
|
||||||
|
|
||||||
|
**Plane/scanout programming is trivial for a compositor.** The primary plane is `PRI_CTL`
|
||||||
|
(enable + pixel format), `PRI_STRIDE`, `PRI_SURF` (surface base — writing it triggers the atomic
|
||||||
|
update), `PRI_OFFSET`; formats include 32-bit BGRX 8:8:8 and 16-bit BGRX 5:6:5 — a direct match for a
|
||||||
|
linear XRGB compositor buffer. Plane registers are double-buffered and latch at vblank via an
|
||||||
|
**arming** write — so a page-flip is "write base + stride + size, then the arming write." This is
|
||||||
|
*exactly* the primitive danos's damage-driven compositor already expresses over GOP/virtio-gpu; the
|
||||||
|
incremental work is "program these display-domain registers," not a new scanout model. The panel
|
||||||
|
fitter (`PF_WIN_POS`/`PF_WIN_SZ`/`PF_CTRL`) can be left disabled for native-resolution scanout;
|
||||||
|
Skylake+ replaces it with a shared pipe-scaler (`PS_CTRL`).
|
||||||
|
|
||||||
|
**Smallest useful target:** eDP (DDI A / transcoder-EDP) or a single DP output at native resolution,
|
||||||
|
panel fitter off, plane in 32bpp XRGB. That is: GMBUS + I²C-over-AUX EDID/DPCD, one fixed-rate or
|
||||||
|
WRPLL config, the ~20-step enable sequence, the software CR/EQ loop, and `PRI_*` plane setup with
|
||||||
|
`PRI_SURF`-write flips. Out of scope: 3D, media, tiling, RC6/power-gating, PSR, audio.
|
||||||
|
|
||||||
|
## Memory and scanout
|
||||||
|
|
||||||
|
This is where Intel's *architecture* — not just its docs — makes the job smaller, and it is the
|
||||||
|
biggest single simplification versus a discrete GPU.
|
||||||
|
|
||||||
|
- **No VRAM.** Intel iGPUs have a unified memory architecture; the display scans out of ordinary
|
||||||
|
**system RAM** addressed through the **Global GTT (GGTT)**. The only way to give the GPU memory is
|
||||||
|
to bind system pages into the GGTT
|
||||||
|
([i915/GEM crashcourse](https://blog.ffwll.ch/2012/10/i915gem-crashcourse.html)).
|
||||||
|
- **The plane surface register is a GGTT offset**, not a raw physical address — the display walks the
|
||||||
|
GGTT to fetch pixels, so a scanout buffer must be GGTT-mapped (global, not per-process). libgfxinit
|
||||||
|
writes the framebuffer offset straight into `DSPSURF`/`PLANE_SURF` masked to 4 KB.
|
||||||
|
- **Linear (untiled) scanout is a first-class supported mode** — the plane's tiling field value 0 is
|
||||||
|
Linear. No X/Y/Yf tiling engine is needed for a display-only driver. (UEFI GOP itself hands off a
|
||||||
|
linear framebuffer the plane is already scanning.)
|
||||||
|
- **No memory manager.** You need only (1) some contiguous-ish system pages and (2) GGTT PTEs
|
||||||
|
pointing at them (`physical_addr | valid_bit` — the GGTT is a flat single-level array of PTEs in
|
||||||
|
the `GTTMMADR` MMIO BAR), then program the plane. **No GEM/TTM/PPGTT/GuC.** coreboot's native-init
|
||||||
|
literally does `for(i…) WRITE32(base + i*inc | 1, (i*4) | 1)`.
|
||||||
|
- **"Stolen memory"** (GSM/DSM) is firmware-reserved system RAM where the firmware places the GGTT
|
||||||
|
itself and the boot framebuffer. A driver is not obligated to keep scanout there — it can rebind
|
||||||
|
GGTT entries to its own pages. Stolen memory matters mainly for *inheriting* the GOP framebuffer at
|
||||||
|
handoff.
|
||||||
|
|
||||||
|
**The contrast with NVIDIA is stark.** On a discrete GPU the scanout surface must live in **VRAM**
|
||||||
|
(nouveau always pins scanout to VRAM), CPU access goes through the **BAR1** aperture (which on
|
||||||
|
consumer cards can be far smaller than total VRAM unless Resizable BAR is on), and you need a
|
||||||
|
contiguous aligned VRAM allocator plus a BAR1 mapping. The Intel iGPU path **eliminates all of that**
|
||||||
|
— scanout is plain system RAM, and a userspace compositor can write the framebuffer pages directly
|
||||||
|
(as danos already does with the GOP WC framebuffer).
|
||||||
|
|
||||||
|
Because danos boots via GOP, an Intel driver attaches to a display whose **GGTT is already populated
|
||||||
|
and whose plane is already scanning a linear framebuffer at native resolution.** A minimal driver can
|
||||||
|
reuse that live mapping and reprogram the running plane rather than come up from cold — the same
|
||||||
|
"attach to a live display" advantage the NVIDIA doc identifies, but with a far smaller register
|
||||||
|
surface and no firmware wall. *(Low-confidence, per-target details to pin from the specific gen's
|
||||||
|
PRM: GGTT PTE size — 4-byte pre-gen8 vs 8-byte gen8+ — the `GTTMMADR`/aperture BAR layout, surface
|
||||||
|
alignment — 4 KB floor but some gens/tilings want 256 KB — and whether the display's GGTT-mediated
|
||||||
|
DMA sits before or after danos's M16 IOMMU on the target platform.)*
|
||||||
|
|
||||||
|
## Firmware
|
||||||
|
|
||||||
|
A minimal display-only Intel driver is **effectively firmware-free — more so than NVIDIA.**
|
||||||
|
|
||||||
|
- **DMC (Display Microcontroller, "CSR", Skylake+) is NOT required for mode-set or scanout.** Its
|
||||||
|
sole job is saving/restoring display-engine registers across DC5/DC6 low-power idle. Absent, i915
|
||||||
|
prints *"Failed to load DMC firmware … Disabling runtime power management"* and the display
|
||||||
|
mode-sets and scans out normally — you lose only the deep display idle states, not output
|
||||||
|
([intel_dmc.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dmc.c);
|
||||||
|
corroborated by multiple distro bug threads). *(A source-level `HAS_DMC` early-return citation would
|
||||||
|
strengthen this beyond distro testimony, but the conclusion is well-supported.)*
|
||||||
|
- **Pre-Skylake parts have no display microcontroller at all** yet perform full mode-set (and even
|
||||||
|
Panel Self Refresh). This confirms the display engine is fundamentally CPU/MMIO-driven; the
|
||||||
|
microcontroller is an add-on for autonomous idling, not a prerequisite for lighting a panel.
|
||||||
|
Targeting a pre-Skylake or DMC-optional generation sidesteps the question entirely.
|
||||||
|
- **GuC and HuC are render/media microcontrollers on the GT side** — GuC schedules the render engines,
|
||||||
|
HuC assists HEVC/H.265 codec (plus later HDCP/PXP/GSC). Neither is in the scanout path; a
|
||||||
|
display-only driver never loads them
|
||||||
|
([kernel.org microcontrollers](https://docs.kernel.org/gpu/i915.html)).
|
||||||
|
- **PSR firmware lives on the panel**, not in the OS — a minimal driver simply doesn't enable PSR.
|
||||||
|
- **Type-C/TCSS (Ice Lake+) firmware** (PMC/IOM/PHY) is part of platform BIOS/coreboot init and the
|
||||||
|
hardware, *not* a signed blob the display driver loads at runtime. A driver attaching to an
|
||||||
|
already-lit GOP connector, or targeting classic DDI ports, avoids it. *(Cold DP-alt-mode changes
|
||||||
|
from a userspace driver on modern TCSS platforms were not traced to primary source — flagged.)*
|
||||||
|
|
||||||
|
There is **no signed-firmware wall over the Intel GPU at all** on the display path. This is the
|
||||||
|
architectural opposite of NVIDIA's mandatory, unsignable, ABI-unstable GSP — which even on the
|
||||||
|
near-side "direct" display path is a permanent maintenance liability for anything beyond scanout.
|
||||||
|
|
||||||
|
## Licensing
|
||||||
|
|
||||||
|
The situation is *better* than NVIDIA's but still nuanced.
|
||||||
|
|
||||||
|
- **The two best code references are both GPL** — Linux i915 (GPL-2.0) and coreboot libgfxinit
|
||||||
|
(GPL-2.0-or-later). You cannot copy either into a permissively-licensed danos. libgfxinit's WRPLL
|
||||||
|
divider math is itself copied from i915, so it carries the same encumbrance.
|
||||||
|
- **But you don't need to copy code.** The Intel PRM is a *specification*, and a clean-room Zig
|
||||||
|
implementation written from the PRM (using libgfxinit/i915 only to understand behaviour, never to
|
||||||
|
copy) is legitimate — register numbers and bit definitions are functional facts, not copyrightable
|
||||||
|
expression. This is the exact inverse of the NVIDIA case, where no such spec exists and the only
|
||||||
|
guide is the GPL/RE'd code itself.
|
||||||
|
- **A permissive precedent exists: Haiku's `intel_extreme` is MIT-licensed** and was built from
|
||||||
|
Intel's public docs. So if danos wants a permissive license, the model is: implement from the PRM,
|
||||||
|
optionally read MIT Haiku for structure, treat GPL libgfxinit/i915 as documentation-of-last-resort.
|
||||||
|
- **A licensing nuance on the recommended generations:** the "copy the 0BSD PRM code samples" shortcut
|
||||||
|
only applies to Tiger-Lake-era (2021+) PRMs. The Haswell/Broadwell/Skylake PRMs are CC-BY-ND, so
|
||||||
|
their register *facts* are free to implement but there are no code samples to lift.
|
||||||
|
|
||||||
|
As with the NVIDIA doc: danos's userspace-driver-over-IPC model (a driver is a separate process behind
|
||||||
|
a defined protocol) is the cleanest possible license boundary if the project ever chooses to ship a
|
||||||
|
GPL display-driver binary and keep the rest of danos permissive — but that is a boundary judgement
|
||||||
|
wanting real diligence, not a settled fact. The clean-room-from-PRM route avoids the question.
|
||||||
|
|
||||||
|
## Prior art outside Linux
|
||||||
|
|
||||||
|
This is a **real contrast with NVIDIA**, where no one has built a from-scratch native driver outside
|
||||||
|
Linux. For Intel there are **multiple independent, non-Linux, clean-room native modeset
|
||||||
|
implementations** to learn from:
|
||||||
|
|
||||||
|
- **coreboot libgfxinit** — SPARK/Ada, G45/GM45 and Arrandale → Coffee Lake + Apollo Lake (TGL
|
||||||
|
in-tree), the strongest structural reference.
|
||||||
|
- **Haiku `intel_extreme`** — modeset-only (no 2D/3D accel), **MIT-licensed**, i845 through Sandy
|
||||||
|
Bridge solid, newer Gemini/Ice/Tiger Lake in progress but "hit or miss, as the driver lags behind
|
||||||
|
the specs" ([Haiku generations](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html),
|
||||||
|
[Phoronix Sept 2024](https://www.phoronix.com/news/Haiku-OS-September-2024)).
|
||||||
|
- **SerenityOS** — added basic native Intel graphics ([PR #6277](https://github.com/SerenityOS/serenity/pull/6277)),
|
||||||
|
though only for very old ICH7-class hardware.
|
||||||
|
- **managarm** — native Intel G45 support.
|
||||||
|
|
||||||
|
The catch: **every clean-room non-Linux implementation targets old hardware.** A modern Gen12 "Xe"
|
||||||
|
desktop iGPU is beyond all of them; for the very newest parts only GPL i915 covers the registers. So
|
||||||
|
the wealth of prior art is real but concentrated below Tiger Lake.
|
||||||
|
|
||||||
|
## The practical desktop caveat
|
||||||
|
|
||||||
|
Before any effort estimate is trusted, three hardware realities — the honest reason "Intel is easier"
|
||||||
|
does **not** automatically mean "it'll light up the reader's monitor":
|
||||||
|
|
||||||
|
1. **Muxing / cabling.** On a desktop with a discrete RTX 3060, the monitor is almost certainly
|
||||||
|
plugged into the *card's* outputs, not the motherboard's. An iGPU driver would light a
|
||||||
|
**different, currently-dark** output. To see danos on Intel the reader would have to physically
|
||||||
|
move the cable to a motherboard video port **and** likely enable the iGPU / "IGD Multi-Monitor" in
|
||||||
|
BIOS. Intel-first probably does **not** light the current display without re-cabling.
|
||||||
|
2. **No iGPU at all.** Intel **F-SKU** desktop chips (i5-9400F, i5-12400F, i5-13400F, i7-13700KF, …)
|
||||||
|
ship the graphics **fused off** and cannot be re-enabled. These are extremely common in
|
||||||
|
budget/mid gaming builds paired with an RTX 3060. On an F-SKU (or an X-series HEDT part) the
|
||||||
|
Intel-iGPU path is a **non-starter** regardless of cabling.
|
||||||
|
3. **Generation coverage.** If the CPU *is* a recent non-F part, its iGPU may be Gen12 Xe (Alder/
|
||||||
|
Raptor Lake), beyond libgfxinit's verified set and beyond most non-Linux prior art — leaving GPL
|
||||||
|
i915 (or the TGL-class PRM, which covers Alder/Raptor display IP) as the only reference.
|
||||||
|
|
||||||
|
A cleaner path for *learning* without the hardware lottery: an older bare-metal Intel box (Haswell/
|
||||||
|
Skylake NUC or laptop) whose panel is natively on the iGPU. Note QEMU does **not** emulate an Intel
|
||||||
|
iGPU display engine, so a VM cannot exercise a real Intel modeset path — virtio-gpu (already working)
|
||||||
|
is the VM answer.
|
||||||
|
|
||||||
|
## Alternatives, and the honest Intel-vs-NVIDIA verdict
|
||||||
|
|
||||||
|
| Option | What you get | The tradeoff |
|
||||||
|
|---|---|---|
|
||||||
|
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; no runtime mode change, no hardware vsync, no multihead |
|
||||||
|
| **Intel iGPU, reuse-GOP** | EDID read + plane page-flips on the GOP-set mode | Still bounded to GOP's resolution; but real driver-owned scanout |
|
||||||
|
| **Intel iGPU, full modeset** (this doc) | Runtime modeset, vsync, multihead, from public docs | Tier 2–3 effort; DP link training; per-gen churn; **needs a cable-attached, documented iGPU** |
|
||||||
|
| **Native NVIDIA GA106 direct** ([nvidia-gpus.md](nvidia-gpus.md)) | Same, on the RTX 3060 the monitor is actually plugged into | **Tier 4**; GPL-only reference; DMA channel modeset; de-emphasised legacy path |
|
||||||
|
| **GA106 via GSP/OGKM** | Also unlocks 3D later | Tier 5; unstable version-pinned firmware ABI |
|
||||||
|
|
||||||
|
**The verdict for *this reader* (RTX 3060 box):** For pure "see danos on my screen," **NVIDIA-direct
|
||||||
|
is paradoxically the more relevant path**, because the monitor is already cabled to the 3060 and GOP
|
||||||
|
already drives it — a native NVIDIA driver reprograms *that* live display. An Intel driver, however
|
||||||
|
much easier to *write*, likely lights a dark motherboard port the reader isn't looking at, or hits an
|
||||||
|
F-SKU with no iGPU.
|
||||||
|
|
||||||
|
**The verdict for *learning display bring-up*:** **Intel wins decisively.** Public register PRMs, four
|
||||||
|
independent open reference drivers, an MIT precedent (Haiku), a compact formally-analysed blueprint
|
||||||
|
(libgfxinit), no signed-firmware wall, no VRAM/BAR memory manager, and a legitimate permissive
|
||||||
|
clean-room path. It reaches "first pixel" far faster than the NVIDIA native path — *on hardware that
|
||||||
|
actually has a cable-attached, documented Intel iGPU.* Those two goals — "run on my machine" and
|
||||||
|
"learn the craft" — point at different silicon, and that is the honest bottom line.
|
||||||
|
|
||||||
|
## "First light" milestones — a danos `.scanout` service
|
||||||
|
|
||||||
|
Framed as a danos `.scanout` service (like the virtio-gpu and proposed NVIDIA ones), inheriting the
|
||||||
|
GOP-initialized display — no firmware, no cold POST:
|
||||||
|
|
||||||
|
1. **PCI/BAR bring-up** — enumerate the iGPU, map its MMIO BAR (`GTTMMADR` + register block) and the
|
||||||
|
aperture BAR via danos MMIO grants; confirm the display engine is GOP-live.
|
||||||
|
2. **EDID** — implement GMBUS DDC (`0x50`) and DP AUX; read + parse the panel EDID and DPCD caps.
|
||||||
|
*(Smallest self-contained, gen-invariant milestone — a good first commit.)*
|
||||||
|
3. **First pixel = reprogram, don't re-modeset** — with GOP's mode and GGTT mapping inherited,
|
||||||
|
reprogram the running plane (`PRI_CTL`/`PRI_STRIDE`/`PRI_SURF`, linear, 32bpp XRGB) to point at a
|
||||||
|
danos-owned system-RAM buffer; prove a page-flip via the `PRI_SURF` arming write on the *current*
|
||||||
|
mode before changing timings. This defers the entire DPLL/DDI/transcoder/link-training surface —
|
||||||
|
the hardest, most gen-specific ~70% of the work.
|
||||||
|
4. **GGTT ownership** — write your own GGTT PTEs (via an MMIO grant to `GTTMMADR`) pointing at
|
||||||
|
compositor-owned pages, for double-buffered damage-driven present.
|
||||||
|
5. **Wire into the compositor `.scanout` backend** (`attach_scanout`); add vsync via the display
|
||||||
|
vblank interrupt (IRQ-as-IPC).
|
||||||
|
6. **Full mode-set** (the hard, gen-specific step) — for one chosen generation (Haswell/Broadwell or
|
||||||
|
Skylake): WRPLL/DPLL programming, the ~20-step DDI/transcoder/pipe enable sequence, panel power
|
||||||
|
sequencing for eDP (`PP_CONTROL`/`PP_ON_DELAYS`/`PP_OFF_DELAYS` — a common black-screen pitfall).
|
||||||
|
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||||
|
software CR/EQ state machine over AUX. TMDS/HDMI avoids it; a live eDP panel avoids most of it.
|
||||||
|
8. **Multihead**, then optionally a second generation once one is solid.
|
||||||
|
|
||||||
|
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||||
|
working display, exactly the resilience v2 already provides via re-attach.
|
||||||
|
|
||||||
|
## Reading list
|
||||||
|
|
||||||
|
**Native reference — coreboot libgfxinit (GPL-2.0-or-later, SPARK/Ada):**
|
||||||
|
- `common/hw-gfx-gma.adb` — `Enable_Output`, the end-to-end modeset state machine.
|
||||||
|
- `common/hw-gfx-dp_training.adb` — the complete generic DP link-training CR/EQ loops.
|
||||||
|
- `common/hw-gfx-gma-pipe_setup.adb` — plane/pipe/scaler + `DSPSURF`/`DSPSTRIDE`/`DSPCNTR` scanout.
|
||||||
|
- `common/hw-gfx-gma-transcoder.adb` — timing generator; `common/hw-gfx-edid.adb`,
|
||||||
|
`hw-gfx-gma-i2c.adb`, `hw-gfx-dp_aux_ch.adb` — EDID/DDC/AUX; `hw-gfx-gma-registers.ads` — offsets.
|
||||||
|
- `common/haswell*/`, `skylake/`, `tigerlake/` — the per-gen PLL/PHY/buffer-translation backends.
|
||||||
|
|
||||||
|
**Vendor register specs — Intel OSRC PRMs:**
|
||||||
|
- [Broadwell Vol 11: Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf)
|
||||||
|
(CC-BY-ND) — the recommended Haswell/Broadwell-class enable sequences, plane, panel fitter.
|
||||||
|
- [Tiger Lake Vol 12: Display Engine](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)
|
||||||
|
(code samples 0BSD) — the most complete modern reference incl. PLL/voltage-swing value tables.
|
||||||
|
- [DG2/Arc Vol 12: Display Engine](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf)
|
||||||
|
— the newest public display PRM (Gen12.5, 2022).
|
||||||
|
- [Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) (Gen4–Gen9.5) and the
|
||||||
|
[kiwitree mirror](https://kiwitree.net/~lina/intel-gfx-docs/prm/) — stable mirrors.
|
||||||
|
|
||||||
|
**GPL reference-of-last-resort — Linux i915 display:**
|
||||||
|
- `intel_gmbus.c`, `intel_dp_aux.c` — the concrete EDID/DDC and DP-AUX register sequences.
|
||||||
|
- `intel_ddi.c` / `intel_ddi_buf_trans.c`, `intel_cdclk.c`, `intel_dpll_mgr.c` — DDI/CDCLK/PLL;
|
||||||
|
`i9xx_plane.c`, `intel_crtc.c` — plane/pipe; `intel_dp.c` — link training. Huge and modular; a
|
||||||
|
reference to confirm undocumented quirks, not a template.
|
||||||
|
|
||||||
|
**Permissive prior art — Haiku `intel_extreme` (MIT):**
|
||||||
|
- [`src/add-ons/kernel/drivers/graphics/intel_extreme/`](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)
|
||||||
|
— a second independent modeset-only driver; MIT, so structurally readable for a permissive danos.
|
||||||
|
- [generations.html](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html)
|
||||||
|
— the best plain-English per-generation fault-line map.
|
||||||
|
|
||||||
|
## Open questions (unresolved by the survey)
|
||||||
|
|
||||||
|
- **Does the target machine have a usable, cable-attached iGPU at all?** F-SKU check, CPU generation,
|
||||||
|
and monitor cabling must be resolved before any effort estimate is trusted (see
|
||||||
|
[practical caveat](#the-practical-desktop-caveat)).
|
||||||
|
- **Does danos even need native mode-*setting*, or only plane/scanout control on the GOP-set mode?**
|
||||||
|
If runtime mode changes aren't required, the driver collapses to EDID + plane page-flips, dropping
|
||||||
|
the DPLL/DDI/link-training ~70% of the work.
|
||||||
|
- **GGTT vs raw physical:** confirm from the exact target-gen PRM that `PLANE_SURF` is interpreted as
|
||||||
|
a GGTT graphics address (well-established, but per-gen confirmation advisable), and the PTE size /
|
||||||
|
`GTTMMADR` / aperture layout for writing GGTT entries.
|
||||||
|
- **Reuse the firmware/GOP GGTT + framebuffer, or install your own GGTT entries?** The latter (needed
|
||||||
|
for double-buffering) means writing GGTT PTEs from the userspace driver via an MMIO grant.
|
||||||
|
- **eDP panel power sequencing** (`PP_*`, T1–T12 delays) — not covered in this pass and a common
|
||||||
|
black-screen source.
|
||||||
|
- **IOMMU interaction** — whether the display's GGTT-mediated DMA needs IOMMU passthrough for the
|
||||||
|
framebuffer pages under danos's M16 IOMMU, or sits before the IOMMU on the target platform.
|
||||||
|
- **DP link-training / AUX robustness and per-generation register drift** are the dominant *risks* —
|
||||||
|
not documentation scarcity.
|
||||||
|
- **Exact Haswell/BDW MMIO offsets** (commonly cited: GMBUS ~`0xC5100`, `DDI_AUX_CTL_A` ~`0x64010`,
|
||||||
|
`DDI_BUF_CTL_A` ~`0x64000`, `DP_TP_CTL_A` ~`0x64040`) were not extracted verbatim from the PRM —
|
||||||
|
confirm against `i915_reg.h` before coding.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
*Research snapshot; verify against current libgfxinit / i915 source and the specific target
|
||||||
|
generation's PRM before building. Intel's public-PRM coverage and the muxing/F-SKU realities of a
|
||||||
|
given machine both change what is actually achievable.*
|
||||||
@@ -1,210 +0,0 @@
|
|||||||
# M17–M18 execution plan: process lifecycle + device manager
|
|
||||||
|
|
||||||
**Archived — completed 2026-07-13** (every item checked; suite ended 54/54).
|
|
||||||
Kept as the record of how M17–M18 landed; the successor is
|
|
||||||
[m19-m20-plan.md](m19-m20-plan.md).
|
|
||||||
|
|
||||||
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
|
||||||
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
|
||||||
settled in those documents; this file is the build order — one phase at a time,
|
|
||||||
each phase green before the next starts. Delete or archive this file when M18
|
|
||||||
lands.
|
|
||||||
|
|
||||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
|
||||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
|
||||||
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
|
||||||
green phase (no co-author trailers).
|
|
||||||
|
|
||||||
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
|
||||||
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
|
||||||
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
|
||||||
phases are all green it is **auto-merged into `main`**; branches are kept after
|
|
||||||
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
|
||||||
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
|
||||||
run the existing QEMU suite green before any new work starts.
|
|
||||||
|
|
||||||
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
|
||||||
only when its definition of green holds.
|
|
||||||
|
|
||||||
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
|
||||||
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
|
||||||
suite green from the worktree (48/48, 2026-07-12).
|
|
||||||
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
|
||||||
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
|
||||||
test; suite 49/49)
|
|
||||||
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
|
||||||
the notification; `process_exit_reason` supervisor-gated;
|
|
||||||
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
|
||||||
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
|
||||||
bounded ref-counted table, publish on every death;
|
|
||||||
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
|
||||||
the owner's death; `vfs-client-death` test; suite 50/50)
|
|
||||||
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
|
||||||
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
|
||||||
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
|
||||||
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
|
||||||
scenario; suite 51/51)
|
|
||||||
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
|
||||||
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
|
||||||
(device-manager-protocol module; the manager as a harness service:
|
|
||||||
supervised spawns, hello deadline via timer sweep, restart with
|
|
||||||
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
|
||||||
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
|
||||||
re-proving claim release each respawn; `driver-restart` scenario;
|
|
||||||
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
|
||||||
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
|
||||||
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
|
||||||
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
|
||||||
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
|
||||||
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
|
||||||
scenario proves report → prune → respawn → re-report; suite 53/53)
|
|
||||||
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
|
||||||
endpoint rides as the call's capability; events are the same structs the
|
|
||||||
buses send); device-list first client; protocol capped at the kernel's
|
|
||||||
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
|
||||||
sheared concurrent serial lines and was the scenario-flake root cause;
|
|
||||||
`device-list` scenario; suite 54/54)
|
|
||||||
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## M17.1 — the kernel releases a dead process's claims
|
|
||||||
|
|
||||||
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
|
||||||
|
|
||||||
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
|
||||||
every `claimed[]` slot holding this task id.
|
|
||||||
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
|
||||||
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
|
||||||
slot in after them, before the exit notification).
|
|
||||||
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
|
||||||
module) and release those by owner in the same pass.
|
|
||||||
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
|
||||||
|
|
||||||
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
|
||||||
device, is killed, is respawned, and claims the same device again successfully;
|
|
||||||
assert both claims in the serial log. Kernel-side unit coverage in
|
|
||||||
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
|
||||||
release one owner, verify exactly its claims freed).
|
|
||||||
|
|
||||||
## M17.2 — exit reasons
|
|
||||||
|
|
||||||
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
|
||||||
illegal_instruction, arithmetic_fault, killed).
|
|
||||||
- Kernel: record the reason at every death site — clean exit path, each fault
|
|
||||||
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
|
||||||
reused, so a small ring keyed by id is enough).
|
|
||||||
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
|
||||||
the recorded reason or `-ESRCH` once evicted.
|
|
||||||
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
|
||||||
- Docs: remove the no-exit-status bullet from process-management.md.
|
|
||||||
|
|
||||||
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
|
||||||
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
|
||||||
three reasons.
|
|
||||||
|
|
||||||
## M17.3 — published exit events
|
|
||||||
|
|
||||||
- Kernel: bounded subscriber table (endpoints); new system call
|
|
||||||
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
|
||||||
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
|
||||||
path already uses.
|
|
||||||
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
|
||||||
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
|
||||||
by that task id (badges already are task ids). Log the release.
|
|
||||||
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
|
||||||
badge encoding).
|
|
||||||
|
|
||||||
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
|
||||||
killed without closing; assert the VFS logs the handle release and its open-handle
|
|
||||||
count returns to baseline.
|
|
||||||
|
|
||||||
## M17.4 — signals and the service harness
|
|
||||||
|
|
||||||
- Kernel: per-task pending mask + bound endpoint; system calls
|
|
||||||
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
|
||||||
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
|
||||||
signals with no bound endpoint pend silently.
|
|
||||||
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
|
||||||
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
|
||||||
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
|
||||||
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
|
||||||
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
|
||||||
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
|
||||||
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
|
||||||
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
|
||||||
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
|
||||||
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
|
||||||
gets for free later.
|
|
||||||
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
|
||||||
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
|
||||||
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
|
||||||
`ping` automatically. Define the reserved `ping` request encoding here and
|
|
||||||
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
|
||||||
existing protocols).
|
|
||||||
- Convert one existing service (input-source or hpet) to the harness as proof it
|
|
||||||
subtracts code rather than adding it.
|
|
||||||
|
|
||||||
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
|
||||||
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
|
||||||
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
|
||||||
deadline with reason `killed`.
|
|
||||||
|
|
||||||
## M18.1 — device-manager protocol: hello + restart policy
|
|
||||||
|
|
||||||
- New `system/services/device-manager/device-manager-protocol.zig` module
|
|
||||||
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
|
||||||
reserved fields.
|
|
||||||
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
|
||||||
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
|
||||||
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
|
||||||
restart vs not.
|
|
||||||
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
|
||||||
conversion is mechanical; otherwise they keep working unconverted (the manager
|
|
||||||
only enforces hello on drivers spawned with an assignment).
|
|
||||||
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
|
||||||
|
|
||||||
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
|
||||||
argv flag to fault after hello on its first run; assert: fault, exit reason
|
|
||||||
recorded, manager respawns with backoff, second run claims the controller
|
|
||||||
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
|
||||||
faults (a tiny test driver, not xhci).
|
|
||||||
|
|
||||||
## M18.2 — bus tree reports
|
|
||||||
|
|
||||||
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
|
||||||
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
|
||||||
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
|
||||||
report one `child_added` per connected port with speed + port number as
|
|
||||||
identity. **No transfer rings, no descriptors** — reading device/interface
|
|
||||||
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
|
||||||
track, not this plan.
|
|
||||||
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
|
||||||
`child_removed`) when a bus driver dies; assert re-report on restart.
|
|
||||||
|
|
||||||
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
|
||||||
`child_added` events reach the manager and appear in its tree dump; kill the
|
|
||||||
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
|
||||||
|
|
||||||
## M18.3 — the application surface
|
|
||||||
|
|
||||||
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
|
||||||
events, input-service pattern).
|
|
||||||
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
|
||||||
becomes the one answer to "what devices exist" for user space.
|
|
||||||
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
|
||||||
discovery migration, out of this plan.
|
|
||||||
|
|
||||||
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
|
||||||
during a driver restart the subscribing client logs remove + add events.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
|
||||||
driver, acpi service, retiring the kernel scan), USB control transfers +
|
|
||||||
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
|
||||||
(needs a console), job control.
|
|
||||||
@@ -1,196 +0,0 @@
|
|||||||
# M19–M20 execution plan: discovery migration
|
|
||||||
|
|
||||||
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
|
||||||
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
|
||||||
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
|
||||||
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
|
||||||
next; this file is the build order and the checklist.
|
|
||||||
|
|
||||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
|
||||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
|
||||||
one), and the relevant design doc updated. Commit per green phase (no co-author
|
|
||||||
trailers). The full suite is the regression net — the existing
|
|
||||||
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
|
||||||
green *through* the migration, which is the whole point: the system must not be
|
|
||||||
able to tell who enumerated it.
|
|
||||||
|
|
||||||
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
|
||||||
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
|
||||||
branch is green; keep branches; push everything.
|
|
||||||
|
|
||||||
## Settled decisions (2026-07-13 — veto before the loop starts)
|
|
||||||
|
|
||||||
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
|
||||||
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
|
||||||
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
|
||||||
power tests prove it), and MCFG (the host bridge node). What retires is
|
|
||||||
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
|
||||||
namespace walk that builds device nodes (M20.3). The AML module stays a
|
|
||||||
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
|
||||||
service (for everything else) — same source, two builds, no fork.
|
|
||||||
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
|
||||||
PCI functions carry BAR resources, and containment demands the bridge own
|
|
||||||
windows that cover them. The apertures are derived kernel-side from the
|
|
||||||
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
|
||||||
mechanical, AML-free, and available at boot regardless of what later moved
|
|
||||||
to user space. (The bridge today carries only ECAM + bus range; this is the
|
|
||||||
prerequisite M19.0 exists for.)
|
|
||||||
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
|
||||||
with identical (parent, class, resources) returns the existing id instead
|
|
||||||
of appending. The kernel table has no unregister, so without this a
|
|
||||||
restarted registering bus would duplicate its children on every respawn —
|
|
||||||
idempotence makes restart-and-re-report safe for every future bus, not just
|
|
||||||
PCI.
|
|
||||||
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
|
||||||
field (the kernel-registered id, `no_device` for unregistered leaves like
|
|
||||||
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
|
||||||
identity (the class triple) instead of the manager's boot-time snapshot —
|
|
||||||
the snapshot match remains only for what the kernel still seeds. One flip
|
|
||||||
phase changes both sides at once so no device is ever matched twice.
|
|
||||||
5. **The acpi service's authority is one node.** The kernel publishes an
|
|
||||||
`acpi-tables` device: memory resources covering the table blobs plus a
|
|
||||||
broad `io_port` resource — the documented trust grant to exactly one
|
|
||||||
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
|
||||||
io_read/io_write calls already exist). The service claims it, maps the
|
|
||||||
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
|
||||||
`mmio_map` + `io_read`/`io_write`.
|
|
||||||
6. **Both new processes are protocol drivers** under the manager: hello,
|
|
||||||
supervision, restart with backoff — all inherited from M18.1 for free.
|
|
||||||
Registration idempotence (decision 3) is what makes their restarts sound.
|
|
||||||
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
|
||||||
everything at and above the device-manager protocol — descriptors,
|
|
||||||
containment, reports, matching, supervision — and none of it may become
|
|
||||||
x86-specific. Discovery is one swappable process per firmware: the acpi
|
|
||||||
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
|
||||||
`devicetree-blob` node, reports children from the flattened device tree —
|
|
||||||
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
|
||||||
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
|
||||||
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
|
||||||
can never take down the supervisor. Two consequences recorded now:
|
|
||||||
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
|
||||||
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
|
||||||
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
|
||||||
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
|
||||||
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
|
||||||
both services exist as placeholders (system/services/acpi, system/services/
|
|
||||||
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
|
||||||
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
|
||||||
in M20.3 and never learn which firmware it is on.
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
- [x] **M19.0** — prerequisites (bridge apertures from the memory map's
|
|
||||||
*gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB,
|
|
||||||
caught by the new every-BAR-contained assert in `discovery`; idempotent
|
|
||||||
`device_register` proven in `bus`; `ChildAdded.device_id`;
|
|
||||||
m17-m18-plan.md archived; suite 54/54).
|
|
||||||
- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM
|
|
||||||
through its grant, brute-force walk with the multifunction rule; the
|
|
||||||
manager matches pci_host_bridge → pci-bus per device with the full
|
|
||||||
protocol contract; `pci-scan` builds its expected marker from the
|
|
||||||
kernel's own count — equivalence on the first run; suite 55/55).
|
|
||||||
- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the
|
|
||||||
kernel's addBars so dedupe returns the kernel's node ids during
|
|
||||||
coexistence; the bridge gained the io_port aperture I/O BARs need;
|
|
||||||
reports carry the registered device_id; pci-scan drills a forced restart
|
|
||||||
and asserts the PCI node count never grows — plus harness hardening: a
|
|
||||||
failing case now preserves its serial as <case>-failed-serial.log, and
|
|
||||||
the heavy scenarios run at 150s; suite 55/55).
|
|
||||||
- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all
|
|
||||||
deleted (bridge node stays); manager matches PCI drivers from reported
|
|
||||||
identity, deduped by registered id. Surfaced and fixed a real SMP race the
|
|
||||||
flip created — ring-3 device_register made the broker table concurrent, so
|
|
||||||
mmio_map's lock-free read intermittently tore hpet's resource length
|
|
||||||
(user fault) and overflowed `r.len-1` into a kernel panic; now the broker
|
|
||||||
read is under the big lock and the arithmetic is guarded, and pci-bus
|
|
||||||
skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart
|
|
||||||
hammered 6×).
|
|
||||||
- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13).
|
|
||||||
- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build
|
|
||||||
module compiled into both kernel and service; the kernel publishes the
|
|
||||||
`acpi-tables` node (AML blobs as memory resources, the broad io_port grant,
|
|
||||||
the SCI); the service claims it, maps the blobs, runs the shared parser in
|
|
||||||
ring 3, and self-verifies its Device count against the kernel's (34 = 34,
|
|
||||||
deterministic via argv, no log-scraping); the manager spawns `discovery`
|
|
||||||
at startup. Parse-only touches no hardware. Suite 56/56.
|
|
||||||
|
|
||||||
- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in
|
|
||||||
ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page
|
|
||||||
backs SystemMemory maps so a stray region can't fault it) and registers +
|
|
||||||
reports each present `_HID` device under `acpi-tables`. Containment: the
|
|
||||||
broker's irq check became range-based (len-1 == the old equality) so the
|
|
||||||
node's broad irq window covers children's legacy lines; io ports fall in
|
|
||||||
the broad io grant. ChildAdded gained `hid`. Matching stays off. The
|
|
||||||
`acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse
|
|
||||||
(1 resource) among the reports. Suite 57/57.
|
|
||||||
- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the
|
|
||||||
device-building helpers are retained-but-dead pending a focused sweep,
|
|
||||||
spawned as a task; static tables + `\_S5` + the acpi-tables node stay).
|
|
||||||
The manager matches ps2-bus from ACPI `_HID` reports; the service
|
|
||||||
registers all devices before reporting any (no keyboard-before-mouse
|
|
||||||
race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches
|
|
||||||
its keyboard; `ioport` retargeted to the acpi-tables I/O window (the
|
|
||||||
kernel-built PS/2 node is gone). Suite 58/58.
|
|
||||||
- [x] **merge** `feat/acpi-service` → main, push (merged 2026-07-13) — **discovery migration complete**.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase notes
|
|
||||||
|
|
||||||
**M19.0 apertures:** the boot memory map already crosses the handoff
|
|
||||||
([boot-handoff]), but discovery never sees it today — expect a small
|
|
||||||
pass-through (kernel init hands the map to the platform layer) before the
|
|
||||||
holes computation, which belongs where the bridge node is built
|
|
||||||
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
|
||||||
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
|
||||||
the kernel unit test.
|
|
||||||
|
|
||||||
**M19.1 scanning without owning config access twice:** the driver reads config
|
|
||||||
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
|
||||||
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
|
||||||
no bridge recursion (matches the kernel's current single-segment walk).
|
|
||||||
|
|
||||||
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
|
||||||
restore) is deferred — the BARs' current programmed values and types are
|
|
||||||
enough for containment-checked registration at bring-up; sizing lands with the
|
|
||||||
first driver that needs to *move* a BAR. Log what is registered so the
|
|
||||||
scenario can assert it.
|
|
||||||
|
|
||||||
**M19.3 what the manager still seeds from the snapshot:** everything the
|
|
||||||
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
|
||||||
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
|
||||||
|
|
||||||
**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns
|
|
||||||
`discovery` by its neutral ramdisk name at startup, as an ordinary protocol
|
|
||||||
driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI
|
|
||||||
devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none):
|
|
||||||
firmware *string* identity travels beside the numeric `identity` field until
|
|
||||||
the FDT-driven widening replaces both (decision 7).
|
|
||||||
|
|
||||||
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
|
||||||
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
|
||||||
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
|
||||||
tell it moved — that is the assertion of `acpi-parse`.
|
|
||||||
|
|
||||||
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
|
||||||
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
|
||||||
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
|
||||||
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
|
||||||
|
|
||||||
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
|
||||||
its spawn moves behind the report (the manager's matching handles this once
|
|
||||||
the source flips); the `input` scenario proves the keyboard still types.
|
|
||||||
|
|
||||||
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
|
||||||
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
|
||||||
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
|
||||||
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
|
||||||
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
|
||||||
has the shape of a signal every driver must answer, and it has no consumer
|
|
||||||
until laptop sleep); CPU P/C-states.
|
|
||||||
|
|
||||||
## M21 — ACPI events + system power — DONE
|
|
||||||
|
|
||||||
Built and merged (docs/m21-plan.md, 2026-07-13): the SCI + power button, Notify/GPE
|
|
||||||
dispatch, and orderly shutdown (init's stop cascade into a ring-3 S5 write).
|
|
||||||
See that plan for the phase record.
|
|
||||||
@@ -1,139 +0,0 @@
|
|||||||
# M21 execution plan: ACPI events + system power
|
|
||||||
|
|
||||||
The operational plan for the event side of the acpi service and orderly
|
|
||||||
shutdown — the capstone [m19-m20-plan.md](m19-m20-plan.md) previewed. Same
|
|
||||||
rules as its predecessors: one phase at a time, each green before the next;
|
|
||||||
this file is the build order and the checklist.
|
|
||||||
|
|
||||||
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
|
||||||
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
|
||||||
phase's new one), and the relevant design doc updated. Commit per green phase
|
|
||||||
(no co-author trailers). Failing cases preserve their serial logs
|
|
||||||
(`<case>-failed-serial.log`).
|
|
||||||
|
|
||||||
**Workflow:** dedicated worktree; branch `feat/power-events` off `main`;
|
|
||||||
auto-merge to main when the branch is green; keep the branch; push everything.
|
|
||||||
|
|
||||||
## Settled decisions (2026-07-13, approved)
|
|
||||||
|
|
||||||
1. **S5 is executed by the acpi service from ring 3.** No new syscall: the
|
|
||||||
broad port grant (M20 decision 5) already made this physically possible —
|
|
||||||
the service holds the PM1 control ports in its io grant and derives `_S5`
|
|
||||||
from its own namespace (`aml.sleepState`). Formalizing it adds no
|
|
||||||
authority. The kernel keeps `power.zig` for its own test paths and
|
|
||||||
panic-time use.
|
|
||||||
2. **The power surface is domain-named** (decision 7 of the last plan): a
|
|
||||||
`power-protocol` module + `ServiceId.power = 5`, registered by the acpi
|
|
||||||
service — on ARM, a PSCI/mailbox service registers the same id and
|
|
||||||
subscribers never know the difference. Messages: `subscribe` (endpoint as
|
|
||||||
the call's capability, the input/manager pattern), `shutdown` (accepted
|
|
||||||
only from PID 1 — init), and events published as buffered messages:
|
|
||||||
`power_button`, `lid`, `ac`, `battery`, generic `notify` with a code.
|
|
||||||
3. **The service learns event ports from its own FADT copy**: the kernel adds
|
|
||||||
the FADT as one more memory resource on the acpi-tables node; the service
|
|
||||||
tells it apart from the AML blobs by signature ("FACP" header — the blob
|
|
||||||
resources are header-stripped bytecode and start with no signature). The
|
|
||||||
kernel's own FADT parse is untouched.
|
|
||||||
4. **The acpi service converts to the harness** (`runtime.service.run`):
|
|
||||||
protocol messages (subscribe/shutdown), the SCI notification, and the
|
|
||||||
existing report flow fold into one loop — the shape it was always meant
|
|
||||||
to have.
|
|
||||||
5. **GPE/Notify correctness is proven by host unit tests** (synthetic AML
|
|
||||||
with a Notify inside a method body; aml.zig joins the `zig build test`
|
|
||||||
loop). The QEMU scenario proves the power button — a *fixed* event,
|
|
||||||
deterministically injectable via QMP `system_powerdown` — because QEMU
|
|
||||||
cannot raise GPEs deterministically on this config. Battery/AC/lid and the
|
|
||||||
embedded controller (`_Qxx`) are interface-complete here and validated on
|
|
||||||
real hardware (the laptop) later.
|
|
||||||
|
|
||||||
## Ground truth the phases build on (verified 2026-07-13)
|
|
||||||
|
|
||||||
- `system/devices/power.zig` `shutdown()` is the kernel's S5 write
|
|
||||||
(SLP_TYP|SLP_EN to PM1a/PM1b control); there is no power syscall.
|
|
||||||
- init (`system/services/init/init.zig`) spawns vfs/input/device-manager
|
|
||||||
fire-and-forget — no child ids kept, no signals, no event loop. The whole
|
|
||||||
stop toolkit exists in `runtime.process` (stop/sendSignal/bindSignals).
|
|
||||||
- `test/qemu_test.py` has no QMP channel (serial is a one-way file).
|
|
||||||
- The kernel parses PM1 *control* blocks and SCI_INT from the FADT; the PM1
|
|
||||||
**event** blocks (offsets 56/60, len at 88) and **GPE0/GPE1** blocks
|
|
||||||
(offsets 80/84, lens 92/93) are unparsed — the service reads them from its
|
|
||||||
FADT copy (decision 3).
|
|
||||||
- The acpi-tables node carries the SCI as its only `len == 1` irq resource
|
|
||||||
(the broad window is len 256) — that is how the service finds it to
|
|
||||||
`irqBind`.
|
|
||||||
- `notify_opcode = 0x86` exists in `system/devices/aml/opcodes.zig` but the
|
|
||||||
interpreter never handles it — a GPE `_Lxx` body containing Notify fails
|
|
||||||
evaluation today. Everything else a GPE handler needs (field access,
|
|
||||||
control flow, method calls) is proven by the ring-3 `_STA`/`_CRS` work.
|
|
||||||
- The dead-code sweep (spawned task) also edits `system/devices/acpi.zig`;
|
|
||||||
M21.0 checks whether it landed and rebases before touching that file.
|
|
||||||
|
|
||||||
## Status
|
|
||||||
|
|
||||||
- [x] **M21.0** — baseline (dead-code sweep confirmed landed on main — no
|
|
||||||
acpi.zig conflict; `feat/power-events` cut; QMP channel in the harness:
|
|
||||||
always-on unix socket, client with the capabilities handshake, per-case
|
|
||||||
`qmp_after` hook, and a hook-must-deliver pass gate that the smoke case
|
|
||||||
now proves with a harmless query-status; suite 58/58).
|
|
||||||
- [x] **M21.1** — SCI + the power button (kernel appends the FADT as an
|
|
||||||
acpi-tables memory resource, tagged by its "FACP" header; `power-protocol`
|
|
||||||
module + `ServiceId.power = 5`; the acpi service converted to
|
|
||||||
`runtime.service.run`, registers `.power`, reads PM1 event/control + GPE
|
|
||||||
ports from its FADT copy, enables ACPI mode if SCI_EN is clear, binds the
|
|
||||||
SCI (the len-1 irq), sets PWRBTN_EN; the SCI handler clears PM1_STS,
|
|
||||||
logs `power: button pressed`, publishes `power_button`, acks. Scenario
|
|
||||||
`power-button` injects a real `system_powerdown` via QMP; initial-ramdisk
|
|
||||||
timeout 30→60s for the service's added boot work; suite 59/59).
|
|
||||||
- [x] **M21.2** — Notify + GPE dispatch (interpreter handles `notify_opcode`
|
|
||||||
into a bounded queue, cleared per-evaluate, drained via
|
|
||||||
`takeNotifications`; the service walks GPE status/enable bytes, evaluates
|
|
||||||
`\_GPE._Lxx`/`_Exx` per active bit, maps notified nodes to events
|
|
||||||
(battery/ac/lid/generic), clears GPE_STS write-1, acks. EC `_Qxx` out.
|
|
||||||
Host unit test with hand-encoded AML proves the queue; aml.zig joined the
|
|
||||||
`zig build test` loop. QEMU raises no GPEs — suite is regression net,
|
|
||||||
59/59).
|
|
||||||
- [x] **M21.3** — orderly shutdown (init supervises its children on one
|
|
||||||
endpoint that also carries signals, power events, and a re-arming
|
|
||||||
heartbeat timer; on `power_button` or a `terminate` signal it logs
|
|
||||||
`init: shutting down`, runs `stop(child, 2000, endpoint)` in reverse
|
|
||||||
order, then requests `.power` shutdown; the acpi service honors shutdown
|
|
||||||
from a subscriber — init is the one subscriber, a soft gate that survives
|
|
||||||
testing where PID 1 isn't init — and writes SLP_TYP|SLP_EN from ring 3.
|
|
||||||
`orderly-shutdown` scenario proves button → shutting-down → S5 → QEMU
|
|
||||||
exit; suite 60/60).
|
|
||||||
- [ ] **merge** `feat/power-events` → main, push, keep the branch — **loop
|
|
||||||
ends here**.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase notes
|
|
||||||
|
|
||||||
**M21.0 QMP:** open the unix socket after Popen, complete the
|
|
||||||
`qmp_capabilities` handshake, then send the hook's command (for these
|
|
||||||
scenarios: `{"execute": "system_powerdown"}`). The socket is additive — no
|
|
||||||
existing case may notice it. Note e3fe3f3 recently reworked how the harness
|
|
||||||
boots; adapt to its current shape rather than the pre-rework description.
|
|
||||||
|
|
||||||
**M21.1 SCI details:** PM1_STS is at the event block base (write-1-to-clear);
|
|
||||||
PM1_EN at base + block_len/2; PWRBTN bit is 8 in both. If PM1b exists, mirror
|
|
||||||
reads/writes to both blocks. Enable ACPI mode only when SCI_EN (PM1 control
|
|
||||||
bit 0) is clear — OVMF boots may already have it set. The publish path reuses
|
|
||||||
the manager's subscriber table pattern (bounded, drop-on-failed-send).
|
|
||||||
|
|
||||||
**M21.2 GPE walk:** GPE0_STS bytes live at the GPE0 block base, GPE0_EN in
|
|
||||||
the block's upper half; for a set+enabled bit n, the handler method is
|
|
||||||
`_L%02X` (level) or `_E%02X` (edge) under `\_GPE`. Evaluate, drain the notify
|
|
||||||
queue, clear the status bit, ack. A missing handler method is clear-and-log,
|
|
||||||
not an error.
|
|
||||||
|
|
||||||
**M21.3 ordering:** init subscribes with retries — the acpi service registers
|
|
||||||
`.power` well after init starts. The stop sequence runs vfs last (other
|
|
||||||
services may flush through it). The S5 write mirrors `power.zig`'s
|
|
||||||
`sleepValue` (SLP_TYP bits [12:10], SLP_EN bit 13); if the write returns, log
|
|
||||||
`power: S5 write did not take` so the scenario fails loudly instead of
|
|
||||||
hanging.
|
|
||||||
|
|
||||||
**Explicitly out of scope:** the embedded controller and `_Qxx` queries,
|
|
||||||
battery `_BST`/`_BIF` evaluation beyond the interface stubs, lid/AC on QEMU
|
|
||||||
(no emulation), reboot over the power protocol, S3 sleep, per-device D-states
|
|
||||||
(a future lifecycle-vocabulary extension), thermal zones.
|
|
||||||
@@ -0,0 +1,246 @@
|
|||||||
|
# Native NVIDIA GPU support — feasibility and roadmap
|
||||||
|
|
||||||
|
**Status: research snapshot, not implemented.** This records what a *native* display driver for a
|
||||||
|
real discrete NVIDIA GPU — specifically an **RTX 3060 (Ampere GA106)** — would take, and how it
|
||||||
|
would slot into danos's pluggable scanout architecture. It is a survey of primary sources
|
||||||
|
(NVIDIA's [open-gpu-kernel-modules](https://github.com/NVIDIA/open-gpu-kernel-modules), the Linux
|
||||||
|
[nouveau/nvkm](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/nouveau) driver,
|
||||||
|
NVIDIA's [open-gpu-doc](https://nvidia.github.io/open-gpu-doc/), and
|
||||||
|
[linux-firmware](https://github.com/NVIDIA/linux-firmware)), not an implementation. The NVIDIA
|
||||||
|
driver landscape moves quickly (GSP defaults, firmware ABIs); treat specifics as a mid-decade
|
||||||
|
snapshot and re-verify against current source before building.
|
||||||
|
|
||||||
|
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the
|
||||||
|
v2 model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||||
|
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||||
|
|
||||||
|
## TL;DR
|
||||||
|
|
||||||
|
- A **minimal display-only driver** (EDID + mode-set + framebuffer scanout, **no** 3D/compute)
|
||||||
|
for the RTX 3060 **can and should avoid the GSP entirely**. nouveau has a register-level,
|
||||||
|
CPU-driven display path for Ampere (`nvkm/engine/disp/ga102.c`) that lights up GA106 with no
|
||||||
|
external firmware; the signed-firmware wall gates the **compute/graphics** engines (PGRAPH),
|
||||||
|
**not** the display controller. "GSP is mandatory on Ampere" is true only for NVIDIA's own
|
||||||
|
RM-object route.
|
||||||
|
- **danos's UEFI GOP boot is the single biggest thing in its favour.** The VBIOS/GOP has already
|
||||||
|
run devinit and brought up the display PLLs, so a driver attaches to a **live, initialized**
|
||||||
|
GA106 — no firmware load, no cold-boot POST, no devinit interpreter. You reprogram a running
|
||||||
|
display rather than bring one up from cold.
|
||||||
|
- It is still a **hard, multi-week-to-months expert effort** (effort tier ≈ 4/5) dominated by
|
||||||
|
NVDisplay channel-DMA programming, SOR/head routing, DisplayPort AUX + link training, and the
|
||||||
|
display supervisor handshake. The GSP/RM route is tier 5 (near-infeasible solo).
|
||||||
|
- The **licensing tension is counterintuitive**: the permissively-licensed reference (NVIDIA
|
||||||
|
open-gpu-kernel-modules, MIT/GPLv2) is the **hard GSP path**; the register-level display code
|
||||||
|
you actually want lives in **GPL nouveau**. See [Licensing](#licensing).
|
||||||
|
- The **window is closing**: GA10x (Ampere) is the *last* NVIDIA family with a register-level
|
||||||
|
display path — Ada (RTX 40) deleted its non-GSP display HAL. Targeting Ampere specifically
|
||||||
|
matters.
|
||||||
|
- **Recommendation:** for *this card*, GOP already gives native-resolution scanout with zero GPU
|
||||||
|
code and zero maintenance. A native driver buys only runtime mode changes, hardware
|
||||||
|
vsync/vblank, and multihead. It is justified if that runtime control is a danos goal, or to
|
||||||
|
*learn the craft* — for which an Intel iGPU or a pre-Turing NVIDIA card reaches "first pixel"
|
||||||
|
far faster.
|
||||||
|
|
||||||
|
## The GSP wall, and why display sits on the near side of it
|
||||||
|
|
||||||
|
On Turing and later, NVIDIA split its driver's Resource Manager into a host **CPU-RM** and a
|
||||||
|
**GSP-RM** running on an on-die RISC-V core ("Peregrine"), talking over RPC
|
||||||
|
([LWN 953144](https://lwn.net/Articles/953144/)). The GSP is a *full resource manager*, not a
|
||||||
|
display coprocessor — there is no "display-only" GSP image and no small display RPC subset. Its
|
||||||
|
boot chain is entirely signed and mandatory: a VBIOS-resident **FWSEC-FRTS** app carves a
|
||||||
|
write-protected region (WPR2), a signed **Booter** on the SEC2 falcon loads the GSP bootloader,
|
||||||
|
and that loads **GSP-RM** inside WPR. The firmware ships pre-computed signatures and the driver
|
||||||
|
picks one by an on-chip fuse-version register — **you cannot self-sign**, and there is **no stable
|
||||||
|
firmware ABI** (it is revised every driver release; nouveau and the Rust nova-core driver each pin
|
||||||
|
exactly one version). A GSP driver is a permanent maintenance liability, not a one-time build
|
||||||
|
([LWN 1037379](https://lwn.net/Articles/1037379/),
|
||||||
|
[nova-core cover letter](https://lore.freedesktop.org/nouveau/20250826-nova_firmware-v2-7-93566252fe3a@nvidia.com/T/)).
|
||||||
|
|
||||||
|
**But display doesn't need any of that on Ampere.** `nvkm/engine/disp/ga102.c` dual-dispatches:
|
||||||
|
|
||||||
|
```
|
||||||
|
if (nvkm_gsp_rm(device->gsp)) return r535_disp_new(&ga102_disp, ...); // GSP RPC path
|
||||||
|
return nvkm_disp_new_(&ga102_disp, ...); // direct register path
|
||||||
|
```
|
||||||
|
|
||||||
|
Both branches use the same `ga102_disp` HAL and the same `GA102_DISP_*` class IDs; GSP merely
|
||||||
|
swaps register programming for RPC. GA106 (chipset `0x176`) is wired to `ga102_disp_new` in the
|
||||||
|
device table, identical to GA102/103/104/107. Ampere lit up displays via the **direct** path in
|
||||||
|
Linux 5.11/5.17 — two years before GSP-RM landed (6.7, 2023)
|
||||||
|
([ga102.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/ga102.c),
|
||||||
|
[Phoronix GA106](https://www.phoronix.com/news/Nouveau-NVIDIA-GA106)).
|
||||||
|
|
||||||
|
**Caveat — this is now the legacy path.** As of Linux 6.18, nouveau defaults to GSP on
|
||||||
|
Turing/Ampere; the direct path is a retained, forceable fallback (`nouveau.config=NvGspRm=0`, and
|
||||||
|
automatic when GSP firmware is absent). It is stable and proven, but NVIDIA and nova-core are
|
||||||
|
moving to GSP-only, and **Ada already deleted its non-GSP display HAL**. GA10x is the last family
|
||||||
|
that keeps a register-level display path.
|
||||||
|
|
||||||
|
## What "direct" actually entails
|
||||||
|
|
||||||
|
"Direct" is not "plain register pokes." Only SOR / PLL / DP-link / clock setup is bare MMIO. The
|
||||||
|
**mode-set and scanout themselves flow through the NVDisplay channels — a DMA pushbuffer**:
|
||||||
|
|
||||||
|
- Display classes for Ampere (the C670 family): core `GA102_DISP_CORE_CHANNEL_DMA` (`0xc67d`),
|
||||||
|
window `0xc67e`, window-immediate `0xc67b`, cursor `0xc67a` (headers `clc67d.h` / `clc67e.h` /
|
||||||
|
`clc67a.h` in [open-gpu-doc `classes/display/`](https://github.com/NVIDIA/open-gpu-doc/tree/master/classes/display)).
|
||||||
|
- The core channel needs **instance memory, a RAMHT, DMA objects, and a channel user-MMIO
|
||||||
|
region** ([disp/chan.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/chan.c)).
|
||||||
|
The register-level "plumbing" to allocate/kick a channel is in NVIDIA's GA102 display register
|
||||||
|
manual: `NV_PDISP_FE_CHNCTL_CORE/WIN/CURS`, `NV_PDISP_FE_PBBASE/PBBASEHI`
|
||||||
|
([dev_display_withoffset.ref.txt](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)).
|
||||||
|
- **Mode-set is a method stream** on the core channel: `HEAD_SET_RASTER_*`,
|
||||||
|
`HEAD_SET_PIXEL_CLOCK_FREQUENCY`, `HEAD_SET_CONTROL_OUTPUT_RESOURCE`, `SOR_SET_CONTROL`
|
||||||
|
(protocol select), viewport/scaler, then `UPDATE`. The window channel points at the scanout
|
||||||
|
surface (`SET_CONTEXT_DMA_ISO`, `SET_STORAGE`, `SET_OFFSET`).
|
||||||
|
- After `UPDATE` you must complete the display **supervisor** interrupt handshake (SV1/SV2/SV3).
|
||||||
|
|
||||||
|
**EDID and DisplayPort are a separate subdev you must port.** open-gpu-doc documents *none* of
|
||||||
|
EDID/DDC/AUX. On the direct path you read EDID in-driver via nouveau's `nvkm/subdev/i2c`: bit-bang
|
||||||
|
**DDC/I²C at address `0x50`** (E-DDC `0x30`) for TMDS/HDMI, or native **DP AUX** in `i2c/aux.c`
|
||||||
|
for DisplayPort. DisplayPort **link training** (the `dp.c` `train_cr` / `train_eq` state machine
|
||||||
|
over AUX — clock recovery, lane/rate, voltage-swing/pre-emphasis) is the single hardest and most
|
||||||
|
fragile piece; a DVI/HDMI (TMDS) panel avoids it entirely.
|
||||||
|
|
||||||
|
## The memory floor (smaller than you'd fear)
|
||||||
|
|
||||||
|
Neither route hands you a framebuffer allocator — even GSP-RM does not manage the scanout
|
||||||
|
framebuffer; the driver owns VRAM and merely tells GSP where its page directory is. But
|
||||||
|
display-only is a small fraction of a full GEM/TTM stack:
|
||||||
|
|
||||||
|
- **Pitch-linear (untiled) scanout is allowed** on nv50→Ampere — the window's storage method has a
|
||||||
|
`PITCH` layout mode, so you skip block-linear tiling math
|
||||||
|
([wndwc37e.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/dispnv50/wndwc37e.c)).
|
||||||
|
- The window references its surface through a simple **display context-DMA**
|
||||||
|
(`SET_CONTEXT_DMA_ISO` + a 256-byte-granular `SET_OFFSET = addr>>8`) — a base/limit descriptor,
|
||||||
|
**not** the GPU's 5-level compute page tables. **No full GPU VMM is needed** for scanout.
|
||||||
|
- The surface must live in **VRAM** in practice (nouveau always pins scanout to VRAM). *Open
|
||||||
|
question:* whether GA10x can scan out from a system-memory (GART) surface via a sysmem-target
|
||||||
|
ctxdma — which would let danos skip a VRAM allocator. No source forbids it; nouveau never does
|
||||||
|
it (confidence: medium).
|
||||||
|
- **CPU access** to the framebuffer for compositing goes through **BAR1** (a VRAM aperture); BAR0
|
||||||
|
is the 16 MB register window. BAR1 can be smaller than 12 GB of VRAM unless Resizable BAR maps
|
||||||
|
it all.
|
||||||
|
|
||||||
|
**Net:** you need (1) a contiguous aligned VRAM allocator (256-byte base, pitch a multiple of
|
||||||
|
64 bytes — confirm against the Ampere display refs), (2) a little instmem for the channel
|
||||||
|
pushbuffers + iso ctxdma, (3) a BAR1 CPU mapping. You do **not** need the 5-level VMM, GEM/TTM
|
||||||
|
eviction, or tiling.
|
||||||
|
|
||||||
|
## Licensing
|
||||||
|
|
||||||
|
The tension is the opposite of convenient:
|
||||||
|
|
||||||
|
- **NVIDIA open-gpu-kernel-modules is dual MIT/GPLv2** — usable under MIT, no copyleft on your
|
||||||
|
other code — **but its display logic is the GSP/RM-object route.** Its class headers
|
||||||
|
(`cl0073.h`, `cl2080.h`, `ctrl0073*.h`) are useful, permissive references.
|
||||||
|
- **nouveau is GPLv2**, and the **register-level display sequences you actually want live in
|
||||||
|
nouveau**, not in the MIT code. So the *easy technical path is the GPL-licensed one.* Reading
|
||||||
|
GPL nouveau and reimplementing it in Zig is a derivative-work risk proportional to how closely
|
||||||
|
your code tracks its structure/constants.
|
||||||
|
|
||||||
|
Options: **(a)** accept that the danos NVIDIA display driver is a **GPL component**. danos's
|
||||||
|
userspace-driver-over-IPC model (a driver is a separate process behind a defined protocol, not
|
||||||
|
linked into the kernel) is about the cleanest possible GPL boundary, so the GPL would be contained
|
||||||
|
to that one binary and the rest of danos could keep its own license — but this is a
|
||||||
|
licensing-boundary judgement that wants real diligence, not a settled fact. **(b)** clean-room
|
||||||
|
from *specification* rather than *code*: [envytools](https://envytools.readthedocs.io) + NVIDIA's
|
||||||
|
open-gpu-doc register manuals + the MIT OGKM class headers, treating nouveau as
|
||||||
|
documentation-of-last-resort.
|
||||||
|
|
||||||
|
**Firmware licensing is moot for the direct path** (no firmware is loaded). For completeness: the
|
||||||
|
GSP blobs are marked redistributable under `LICENCE.nvidia`, which permits use by **any
|
||||||
|
OSI-approved open-source OS** (not just Linux), on NVIDIA GPUs, **unmodified**, with **no
|
||||||
|
reverse-engineering of the firmware binary**. The one gate — is danos released under an OSI
|
||||||
|
license? — is only reached on the GSP route, which this doc recommends against for this card.
|
||||||
|
|
||||||
|
## Prior art
|
||||||
|
|
||||||
|
**No one has built a from-scratch native NVIDIA driver outside Linux.** FreeBSD ships
|
||||||
|
`nvidia-drm-kmod`, a *port of NVIDIA's own closed `nvidia-drm.ko`* loading the GSP blob (its old
|
||||||
|
nouveau port was removed). Haiku's NVIDIA support is likewise a *port of OGKM* (GSP, Turing+, very
|
||||||
|
alpha). OpenBSD / DragonFly have neither. Every non-Linux OS that supports modern NVIDIA chose to
|
||||||
|
**wrap NVIDIA's GSP stack** rather than write a native driver. A danos direct-register driver
|
||||||
|
would have exactly one reference implementation — GPL nouveau — and no non-Linux precedent.
|
||||||
|
|
||||||
|
## Alternatives
|
||||||
|
|
||||||
|
| Option | What you get | The tradeoff |
|
||||||
|
|---|---|---|
|
||||||
|
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; **no runtime mode change, no hardware vsync, no multihead** |
|
||||||
|
| **Pre-Turing NVIDIA** (Kepler / early Maxwell) | Direct EVO/disp-core + CRTC/PLL modeset, **no signed firmware, no coprocessor**; mature nouveau reference | Older display class; not this card; only reclocking is firmware-gated |
|
||||||
|
| **Intel iGPU** | **Publicly documented** register interfaces (Intel PRMs); no coprocessor mediating modeset | i915 is huge + generation-specific; write one generation from the PRM |
|
||||||
|
| **Native GA106 direct** (this doc) | Runtime modeset, vsync, multihead on the actual card | Tier-4 effort; GPL reference; DP link training; legacy/de-emphasized path |
|
||||||
|
| **GA106 via GSP/OGKM** | Also unlocks 3D / reclocking later | Tier-5; ~14k-line ante; unstable version-pinned ABI; unprecedented outside Linux |
|
||||||
|
|
||||||
|
## "First light" milestones (direct path, inheriting GOP state)
|
||||||
|
|
||||||
|
Framed as a danos `.scanout` service (like the virtio-gpu driver), taking the direct register path
|
||||||
|
and inheriting the GOP-initialized display — no signed firmware, no devinit, no GSP:
|
||||||
|
|
||||||
|
1. **PCI/BAR bring-up** — enumerate GA106 (`0x176`), map **BAR0** (registers) and **BAR1** (VRAM
|
||||||
|
aperture) via danos MMIO grants; confirm the display engine is GOP-live.
|
||||||
|
2. **VRAM + instmem allocator** — contiguous aligned VRAM for the scanout surface (256-byte base)
|
||||||
|
+ small instmem for pushbuffers / RAMHT / iso ctxdma. No VMM, no TTM.
|
||||||
|
3. **EDID** — port `nvkm/subdev/i2c` DDC (`0x50`) + DP-AUX (`aux.c`); read + parse the panel EDID.
|
||||||
|
4. **Core channel up** — allocate the `0xc67d` core channel as a DMA pushbuffer; stand up the
|
||||||
|
SV1/SV2/SV3 supervisor-interrupt handshake.
|
||||||
|
5. **First pixel = reprogram, don't re-POST** — bind a window (`0xc67e`) at the existing WC
|
||||||
|
framebuffer via `SET_CONTEXT_DMA_ISO` + `SET_OFFSET`, pitch-linear, `UPDATE`; prove you can
|
||||||
|
drive the *current* GOP mode from your own channel before changing anything.
|
||||||
|
6. **Modeset** — push raster timings on a head, route head→SOR→connector, program the pixel-clock
|
||||||
|
PLL, switch to an EDID mode (needs the `clc67d/e` method opcodes from the OGKM headers + the
|
||||||
|
supervisor timing from nouveau `head.c`).
|
||||||
|
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||||
|
`dp.c` `train_cr`/`train_eq` state machine. TMDS/HDMI is far simpler.
|
||||||
|
8. **Wire into the compositor `.scanout` backend** (`attach_scanout`), add vsync via the display
|
||||||
|
interrupt, then multihead.
|
||||||
|
|
||||||
|
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||||
|
working display (exactly the resilience v2 already provides via re-attach).
|
||||||
|
|
||||||
|
## Reading list
|
||||||
|
|
||||||
|
**Direct path — nouveau (GPLv2):**
|
||||||
|
- `nvkm/engine/disp/ga102.c` — the GA10x display HAL + the GSP/non-GSP dispatch.
|
||||||
|
- `nvkm/engine/disp/{head.c, ior.c, dp.c, hdmi.c, chan.c}` — head/SOR routing, DP AUX + link
|
||||||
|
training, channel-DMA plumbing.
|
||||||
|
- `dispnv50/{corec37d.c, corec57d.c, wndwc37e.c, wndwc57e.c, wndwc67e.c, headc37d.c, cursc37a.c}`.
|
||||||
|
- `nvkm/subdev/i2c` (DDC + `aux.c`) for EDID; `nvkm/subdev/bios/init.c` + `devinit/` **only** if
|
||||||
|
you ever have to re-POST (danos's GOP handoff means you shouldn't).
|
||||||
|
|
||||||
|
**Object model / GSP path — NVIDIA OGKM (MIT/GPLv2):** class headers `cl0073.h`, `cl2080.h`,
|
||||||
|
`ctrl0073system.h`, `ctrl0073specific.h`; `src/nvidia/` for RM control sequences.
|
||||||
|
`nvidia-modeset.ko` (NVKMS) is a *policy* layer over RM and can be bypassed entirely.
|
||||||
|
[nova-core](https://lore.freedesktop.org/nouveau/) (Rust) is the forward-looking reference for GSP
|
||||||
|
boot mechanics (falcon signing, queue rings, RPC).
|
||||||
|
|
||||||
|
**Register / method specs — NVIDIA open-gpu-doc:**
|
||||||
|
- [`classes/display/README.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/README.txt)
|
||||||
|
— the channel model + class-to-GPU map (read first).
|
||||||
|
- [`classes/display/clc67d.h`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/clc67d.h)
|
||||||
|
+ `clc67e.h` / `clc67a.h` — the Ampere core/window/cursor mode-set method vocabulary.
|
||||||
|
- [`manuals/ampere/ga102/dev_display_withoffset.ref.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)
|
||||||
|
— `NV_PDISP_FE_*` channel/pushbuffer registers + SOR.
|
||||||
|
- [`DCB`](https://github.com/NVIDIA/open-gpu-doc/tree/master/DCB) — connector→output-resource
|
||||||
|
routing; [`Devinit`](https://github.com/NVIDIA/open-gpu-doc/tree/master/Devinit) +
|
||||||
|
[`BIOS-Information-Table`](https://github.com/NVIDIA/open-gpu-doc/tree/master/BIOS-Information-Table)
|
||||||
|
— VBIOS parsing (bring-up reference; not needed if inheriting GOP).
|
||||||
|
- The 632 KB Volta [`dev_display.ref`](https://download.nvidia.com/open-gpu-doc/Display-Ref-Manuals/1/gv100/dev_display.ref)
|
||||||
|
is the best shot at SOR-DP/AUX register detail the smaller Ampere file omits.
|
||||||
|
|
||||||
|
## Open questions (unresolved by the survey)
|
||||||
|
|
||||||
|
Each needs a direct read of the named nouveau file or experimentation on the actual card:
|
||||||
|
|
||||||
|
- Exact GA106 register/method offsets and PADLINK→SOR→connector wiring (can vary by board vendor).
|
||||||
|
- Whether *any* PLL/devinit re-run is unavoidable vs. fully inherited from GOP.
|
||||||
|
- Whether DisplayPort needs full retraining on takeover, or the GOP-established link can be reused.
|
||||||
|
- The precise SV1/SV2/SV3 supervisor sequence.
|
||||||
|
- Whether a system-memory-target scanout ctxdma could eliminate the VRAM allocator.
|
||||||
|
- The exact `clc67d.h`/`clc67e.h` method opcode numbers (not captured verbatim in the survey).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
*Research snapshot; verify against current nouveau / open-gpu-kernel-modules source before
|
||||||
|
building — NVIDIA's GSP defaults and firmware ABIs change per release.*
|
||||||
+128
@@ -0,0 +1,128 @@
|
|||||||
|
# The power service: events and shutdown
|
||||||
|
|
||||||
|
A laptop lid closes, a battery drains, someone presses the power button — and
|
||||||
|
several parts of the system might care: a session manager dims the screen, a
|
||||||
|
logger notes it, and ultimately *something* has to turn the machine off. None of
|
||||||
|
them owns the hardware that reported the event, and the reporter should not know
|
||||||
|
who is listening. So system power is a **service**: an event source **publishes**
|
||||||
|
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||||
|
privileged caller — init — can ask it to power the machine off. It is the same
|
||||||
|
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||||
|
|
||||||
|
## Why a service, and why it is named for the domain, not the firmware
|
||||||
|
|
||||||
|
Where the events come from is firmware-specific — on x86 they ride the ACPI SCI
|
||||||
|
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||||
|
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||||
|
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||||
|
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||||
|
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||||
|
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||||
|
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||||
|
preserve, carried one layer up into a running-system surface.
|
||||||
|
|
||||||
|
This is why the protocol is `power`, not "ACPI events": naming a cross-firmware
|
||||||
|
surface after one firmware would leak x86 into code the ARM port must reuse
|
||||||
|
unchanged.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||||
|
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||||
|
fields. Three operations:
|
||||||
|
|
||||||
|
| Direction | Operation | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||||
|
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||||
|
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||||
|
|
||||||
|
Events are published, not polled: like the input service, the service holds
|
||||||
|
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||||
|
message, so a slow or dead subscriber can never wedge the source. The event
|
||||||
|
vocabulary is hardware-neutral:
|
||||||
|
|
||||||
|
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||||
|
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||||
|
- `notify` — a device notification that maps to none of the above; its `code`
|
||||||
|
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||||
|
and what happened.
|
||||||
|
|
||||||
|
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||||
|
generic `notify` is fully described without a second round trip.
|
||||||
|
|
||||||
|
**`shutdown` is authority, not information.** It is the only operation that
|
||||||
|
*does* something irreversible, so it is gated: the contract is that only init
|
||||||
|
(PID 1) may request it, because init is the process that has already run the stop
|
||||||
|
sequence over everything else. The acpi service implements this as a **soft
|
||||||
|
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||||
|
init is the one subscriber. That stands in for "only the system supervisor may
|
||||||
|
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||||
|
is not init.
|
||||||
|
|
||||||
|
## Orderly shutdown
|
||||||
|
|
||||||
|
Powering off cleanly is where the power service, the [process
|
||||||
|
lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init
|
||||||
|
already supervises the services it starts; for shutdown it runs **one event loop
|
||||||
|
over one endpoint** that carries three things at once: its children's exit
|
||||||
|
notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||||
|
**power events** it subscribes to — plus a re-arming heartbeat timer proving PID
|
||||||
|
1 is alive. (init subscribes with retries, because the power service registers
|
||||||
|
`.power` well after init starts; a missing power service is not fatal — a
|
||||||
|
`terminate` signal drives the same path.)
|
||||||
|
|
||||||
|
On a `power_button` event or a `terminate` signal, init:
|
||||||
|
|
||||||
|
1. logs that it is shutting down,
|
||||||
|
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||||
|
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||||
|
last (other services may flush through it), each child getting the
|
||||||
|
*terminate → deadline → kill* escalation from
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), and
|
||||||
|
3. requests `.power` `shutdown`.
|
||||||
|
|
||||||
|
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||||
|
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||||
|
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||||
|
the machine off, it logs loudly so a test fails rather than hangs.
|
||||||
|
|
||||||
|
**No new system call was needed for S5.** The broad io_port grant on the
|
||||||
|
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||||
|
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||||
|
could physically already do; formalizing it as a protocol operation added a
|
||||||
|
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||||
|
panic-time poweroff, where no user space is available to ask.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
Two QEMU scenarios exercise the path, both injecting a real ACPI power-button
|
||||||
|
press via QMP `system_powerdown` (there is no other deterministic power event on
|
||||||
|
this config):
|
||||||
|
|
||||||
|
- `power-button` proves the source: the acpi service's SCI handler logs the
|
||||||
|
press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)).
|
||||||
|
- `orderly-shutdown` proves the whole composition: button → init logs shutting
|
||||||
|
down → children stopped → the service enters S5 → QEMU exits. The ordered
|
||||||
|
regex is the proof, and QEMU's self-exit through S5 is the pass.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
Interface-complete but validated on real hardware (the author's laptop) later,
|
||||||
|
because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the
|
||||||
|
interface stubs, lid and AC events, and the embedded controller's `_Qxx`
|
||||||
|
queries. Deliberately out of scope for now: reboot over the power protocol, S3
|
||||||
|
sleep, per-device D-states (a future lifecycle-vocabulary extension, since
|
||||||
|
"suspend" has the shape of a signal every driver must answer and has no consumer
|
||||||
|
until laptop sleep), and thermal zones.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power
|
||||||
|
button fixed event, and GPE/Notify dispatch in the acpi service.
|
||||||
|
- [discovery.md](discovery.md) — why the surface is domain-named, and the
|
||||||
|
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||||
|
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||||
|
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||||
|
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||||
|
its own children.
|
||||||
@@ -17,12 +17,12 @@ was a mistake) without inheriting the mechanism, the API, or the names. The nami
|
|||||||
rule is danos's own and it is strict: plain words that communicate intent
|
rule is danos's own and it is strict: plain words that communicate intent
|
||||||
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: a
|
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||||
**musl-based C layer** (growing out of library/posix) that wires C programs to the
|
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||||
danos runtime — musl's syscall surface retargeted at danos system calls and IPC
|
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||||
protocols (files onto the VFS protocol, `sigaction`/`wait` onto this lifecycle,
|
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||||
sockets onto whatever networking becomes). Ported programs see POSIX; the system
|
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||||
underneath never does.
|
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||||
|
|
||||||
## Why a standard vocabulary
|
## Why a standard vocabulary
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -11,7 +11,7 @@ because the kernel releases a dead process's claims. The `driver-restart` and
|
|||||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||||
end to end. What remains of this document's ladder is scope, not mechanism:
|
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||||
more of the system moved into restartable processes (the discovery migration,
|
more of the system moved into restartable processes (the discovery migration,
|
||||||
[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing:
|
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
|
|||||||
@@ -0,0 +1,227 @@
|
|||||||
|
# System Requirements
|
||||||
|
|
||||||
|
Minimum and recommended hardware for running danos. Every requirement below is
|
||||||
|
grounded in what the current code actually assumes at boot — this is a
|
||||||
|
description of the real target, not an aspirational one.
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
danos targets a **modern UEFI x86-64 PC with ACPI and PCIe**. The practical
|
||||||
|
minimum is:
|
||||||
|
|
||||||
|
- 64-bit x86-64 CPU with SSE2, APIC, and `syscall`/`sysret`
|
||||||
|
- UEFI firmware (no BIOS / legacy boot)
|
||||||
|
- ACPI tables: MADT, MCFG, FADT
|
||||||
|
- PCIe with an ECAM (MMConfig) window
|
||||||
|
- **128 MiB RAM** (target); see [Memory](#memory) for the breakdown
|
||||||
|
- USB via **xHCI only**
|
||||||
|
|
||||||
|
There is no support for legacy BIOS boot, x2APIC, port-IO PCI configuration, or
|
||||||
|
any USB host controller other than xHCI.
|
||||||
|
|
||||||
|
## Plain-language hardware guide
|
||||||
|
|
||||||
|
If you don't want to cross-reference chipset datasheets, here's roughly what era
|
||||||
|
of PC works. These are **guidance based on when the required features became
|
||||||
|
standard**, not a list of tested machines — the authoritative rules are in the
|
||||||
|
technical sections below.
|
||||||
|
|
||||||
|
The feature that sets the floor is **built-in xHCI USB** (danos supports no other
|
||||||
|
USB controller) combined with **UEFI firmware**. Both became standard on
|
||||||
|
mainstream desktops and laptops around **2012**.
|
||||||
|
|
||||||
|
| | Known-good baseline | Comfortable recommendation |
|
||||||
|
|---|---|---|
|
||||||
|
| **Intel** | 3rd-gen Core "Ivy Bridge" (2012) with a 7-series "Panther Point" chipset — Intel's first chipset with xHCI built in | 6th-gen Core "Skylake" (2015) or newer |
|
||||||
|
| **AMD** | A-series "Llano" APU with an A75 FCH (2011) — the industry's first chipset with built-in xHCI | Any AM4 platform, i.e. Ryzen (2017) or newer |
|
||||||
|
|
||||||
|
**AMD is not behind Intel here — it was first.** AMD's A75 FCH shipped with
|
||||||
|
native xHCI in April 2011, about a year *ahead* of Intel's 7-series (2012); AMD
|
||||||
|
was the first vendor to earn USB-IF certification for chipset-level USB 3.0. The
|
||||||
|
two "comfortable recommendation" dates differ only because they name convenient,
|
||||||
|
long-supported product lines (Skylake, Ryzen) — not because of any USB
|
||||||
|
capability gap. Every AMD desktop platform from the A75 FCH (2011) and FM2/AM3+
|
||||||
|
era onward has built-in xHCI, and any of them qualifies as a baseline.
|
||||||
|
|
||||||
|
Older 64-bit machines (e.g. Intel Core 2, Nehalem, Sandy Bridge) meet the CPU
|
||||||
|
requirements but typically **lack built-in xHCI and/or ship with BIOS instead of
|
||||||
|
UEFI**, so they are not supported.
|
||||||
|
|
||||||
|
### Matching your CPU by name
|
||||||
|
|
||||||
|
If you know your chip's marketing name or codename, find it here. Everything from
|
||||||
|
the **Supported** rows down works; the **Too old** row does not.
|
||||||
|
|
||||||
|
**Intel Core** (the "-lake"/"-bridge"/"-well" codenames):
|
||||||
|
|
||||||
|
| Status | Generation | Codename(s) | Year |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Too old | 2nd gen | Sandy Bridge | 2011 |
|
||||||
|
| Supported (baseline) | 3rd gen | Ivy Bridge | 2012 |
|
||||||
|
| Supported | 4th–5th gen | Haswell, Broadwell | 2013–2014 |
|
||||||
|
| **Recommended** | 6th–9th gen | **Skylake**, Kaby Lake, Coffee Lake | 2015–2018 |
|
||||||
|
| Recommended | 10th–11th gen | Comet Lake, Ice Lake, Tiger Lake, Rocket Lake | 2019–2021 |
|
||||||
|
| Recommended | 12th gen+ | Alder Lake, Raptor Lake | 2021–2023 |
|
||||||
|
| Recommended | Core Ultra | Meteor Lake, Arrow Lake, Lunar Lake | 2023+ |
|
||||||
|
|
||||||
|
**AMD:**
|
||||||
|
|
||||||
|
| Status | Family | Codename(s) | Year |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Supported (baseline) | A-series APU (A75/A85 FCH) | Llano, Trinity, Richland, Kaveri | 2011–2014 |
|
||||||
|
| Supported | FX (AM3+) | Bulldozer, Piledriver | 2011–2012 |
|
||||||
|
| **Recommended** | **Ryzen** 1000–5000 (AM4) | Summit/Pinnacle Ridge, Matisse, Vermeer (Zen–Zen 3) | 2017–2020 |
|
||||||
|
| Recommended | Ryzen 7000+ (AM5) | Raphael, Granite Ridge (Zen 4 / Zen 5) | 2022+ |
|
||||||
|
| Recommended | Threadripper / EPYC | Zen and later | 2017+ |
|
||||||
|
|
||||||
|
(These map generations to the era their platforms shipped built-in xHCI + UEFI;
|
||||||
|
they are guidance, not a tested-hardware list.)
|
||||||
|
|
||||||
|
**Two caveats that matter regardless of CPU:**
|
||||||
|
|
||||||
|
- **Firmware must be UEFI.** Many 2011-era machines could do either UEFI or
|
||||||
|
legacy BIOS — danos needs it set to UEFI. There is no BIOS boot path.
|
||||||
|
- **Input is PS/2 only, for now.** danos does not yet support USB
|
||||||
|
keyboards/mice. This is fine on most **laptops** (their built-in keyboards are
|
||||||
|
wired to a PS/2-style i8042 controller) but means a **desktop with only USB
|
||||||
|
ports** currently has no usable keyboard. USB HID input is planned.
|
||||||
|
|
||||||
|
Virtual machines are the easiest way to meet every requirement: QEMU (with OVMF/
|
||||||
|
UEFI, a `qemu-xhci` controller, and the default Q35 machine type), or any
|
||||||
|
hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||||
|
|
||||||
|
## CPU / architecture
|
||||||
|
|
||||||
|
| Requirement | Detail | Source |
|
||||||
|
|---|---|---|
|
||||||
|
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:285`, `boot/efi.zig:418` |
|
||||||
|
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||||
|
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:282`, `trampoline.s:62` |
|
||||||
|
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:59`, `isr.s:169` |
|
||||||
|
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:62`, `apic.zig:414` |
|
||||||
|
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:279`, `apic.zig:84` |
|
||||||
|
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:144` |
|
||||||
|
|
||||||
|
## Firmware / boot
|
||||||
|
|
||||||
|
- **UEFI only.** A custom UEFI application loader is installed to
|
||||||
|
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||||
|
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||||
|
(`build.zig:464`, `boot/efi.zig`)
|
||||||
|
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||||
|
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||||
|
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||||
|
(`efi.zig:578`, `boot-handoff.zig:144`)
|
||||||
|
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||||
|
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||||
|
(`system/devices/acpi.zig:3`)
|
||||||
|
- The loader reads `/system/kernel`, `/system/services/init`, and
|
||||||
|
`/boot/initial-ramdisk.img` off the FAT boot volume. The kernel can boot
|
||||||
|
"kernel-only" without init or the ramdisk. (`efi.zig:14`, `efi.zig:66`)
|
||||||
|
|
||||||
|
## Interrupt controller
|
||||||
|
|
||||||
|
- **Local APIC + I/O APIC required.** I/O APIC base, GSI base, and MADT
|
||||||
|
interrupt-source overrides come from ACPI. (`cpu.zig:365`, `apic.zig:119`)
|
||||||
|
- **MSI supported** — edge-triggered, keyed by vector, no I/O APIC mask cycle.
|
||||||
|
Vector window 33–46, timer on 32, spurious on 47. (`system/kernel/irq.zig:70`,
|
||||||
|
`cpu.zig:397`)
|
||||||
|
- The legacy 8259 PIC is remapped and masked **only if present** (MADT
|
||||||
|
`PCAT_COMPAT`); it is not required. (`apic.zig:103`)
|
||||||
|
|
||||||
|
## PCI / PCIe
|
||||||
|
|
||||||
|
- **PCIe with ECAM (MMConfig) required.** The PCI bus driver maps the host
|
||||||
|
bridge's ECAM window (1 MiB config space per bus) and computes config
|
||||||
|
addresses directly. **There is no legacy CF8/CFC port-IO config path** — the
|
||||||
|
driver bails if the bridge exposes no ECAM window. The ECAM base comes from
|
||||||
|
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:41`, `acpi.zig:6`)
|
||||||
|
|
||||||
|
## USB
|
||||||
|
|
||||||
|
- **xHCI only.** The sole USB driver is `usb-xhci-bus`, and the device manager
|
||||||
|
binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI / OHCI / EHCI exist only
|
||||||
|
as report strings with no driver behind them — **USB 1.x/2.0-only controllers
|
||||||
|
are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||||
|
`system/services/device-manager/device-manager.zig:34`)
|
||||||
|
- USB input (keyboard/mouse over HID) is future work; the current input stack is
|
||||||
|
PS/2. See [Buses & devices](#buses--devices).
|
||||||
|
|
||||||
|
## Timers
|
||||||
|
|
||||||
|
Calibration prefers, in order: (1) CPUID leaf `0x15` TSC frequency, (2) HPET,
|
||||||
|
(3) ACPI PM timer (3.579545 MHz, from FADT), (4) legacy PIT. Any one suffices —
|
||||||
|
HPET/PM-timer/PIT are optional fallbacks when CPUID `0x15` is absent.
|
||||||
|
(`apic.zig:180`)
|
||||||
|
|
||||||
|
- **TSC** — monotonic high-resolution clock.
|
||||||
|
- **LAPIC timer** — scheduler heartbeat, periodic at `timer_hz = 1000 Hz`.
|
||||||
|
(`parameters.zig:39`)
|
||||||
|
|
||||||
|
## Memory
|
||||||
|
|
||||||
|
**Target: 128 MiB RAM.** The system uses 4 KiB pages and a bitmap physical-frame
|
||||||
|
allocator built from the firmware memory map. There is no hardcoded minimum-RAM
|
||||||
|
constant — the allocator only panics if there is no usable region, or none large
|
||||||
|
enough to hold its own bitmap. (`system/kernel/pmm.zig:13`, `pmm.zig:77`)
|
||||||
|
|
||||||
|
Where the budget goes:
|
||||||
|
|
||||||
|
| Consumer | Size | Source |
|
||||||
|
|---|---|---|
|
||||||
|
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:26` |
|
||||||
|
| Kernel stack, per CPU | 16 KiB | `parameters.zig:26` |
|
||||||
|
| IST stack, per CPU | 16 KiB | `parameters.zig:36` |
|
||||||
|
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:32` |
|
||||||
|
| Max concurrent tasks | 32 | `parameters.zig:23` |
|
||||||
|
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:299` |
|
||||||
|
|
||||||
|
The 64 MiB heap cap plus kernel image, per-CPU stacks, task stacks, the frame
|
||||||
|
bitmap, and DMA-contiguous allocations fit comfortably within 128 MiB on a
|
||||||
|
single- or low-core-count machine. Very high core counts (toward the 128-CPU
|
||||||
|
ceiling) add per-CPU stack overhead and push toward more RAM.
|
||||||
|
|
||||||
|
**Note on the 4 GiB physmap:** the loader identity-maps and physmaps the low
|
||||||
|
4 GiB of address space with 2 MiB leaves. This is *virtual address* reach, not a
|
||||||
|
RAM requirement — RAM above 4 GiB simply needs an extra mapping window and is not
|
||||||
|
needed to boot. (`efi.zig:305`)
|
||||||
|
|
||||||
|
Virtual-memory layout (`boot-handoff.zig:47`):
|
||||||
|
|
||||||
|
| Region | Base |
|
||||||
|
|---|---|
|
||||||
|
| User space | `0x0000_7000_0000_0000` |
|
||||||
|
| Kernel heap | `0xFFFF_8000_0000_0000` |
|
||||||
|
| Physmap | `0xFFFF_8800_0000_0000` |
|
||||||
|
| Kernel image | `0xFFFF_FFFF_8000_0000` |
|
||||||
|
|
||||||
|
## Buses & devices
|
||||||
|
|
||||||
|
Buses with real drivers today:
|
||||||
|
|
||||||
|
- **PCIe** via ECAM (`pci-bus`)
|
||||||
|
- **xHCI USB** (`usb-xhci-bus`)
|
||||||
|
- **PS/2** keyboard + mouse (`ps2-bus`) — the current input stack
|
||||||
|
- **Serial UART** (16550/16450), configured from the ACPI SPCR table
|
||||||
|
|
||||||
|
**No storage driver exists yet.** AHCI / NVMe / IDE are named for reporting only;
|
||||||
|
there is no block-device driver. Persistent storage is future work.
|
||||||
|
|
||||||
|
## IOMMU
|
||||||
|
|
||||||
|
**Detection only; enforcement deferred.** The ACPI DMAR table is parsed for the
|
||||||
|
first VT-d DRHD unit and its capabilities are exposed via `PlatformInfo`
|
||||||
|
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||||
|
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||||
|
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||||
|
|
||||||
|
## What is explicitly NOT supported
|
||||||
|
|
||||||
|
- Legacy BIOS / multiboot / limine boot
|
||||||
|
- 32-bit x86
|
||||||
|
- x2APIC
|
||||||
|
- Legacy port-IO (CF8/CFC) PCI configuration
|
||||||
|
- Non-xHCI USB (UHCI / OHCI / EHCI)
|
||||||
|
- Machines without ACPI (no device discovery)
|
||||||
|
- Persistent storage (no AHCI / NVMe / IDE driver yet)
|
||||||
|
- USB HID input (PS/2 only for now)
|
||||||
@@ -27,6 +27,15 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
|||||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||||
a new architecture's UART is what makes the same tests run there.
|
a new architecture's UART is what makes the same tests run there.
|
||||||
|
|
||||||
|
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||||
|
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||||
|
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||||
|
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||||
|
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||||
|
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||||
|
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||||
|
so a serial-enabled image is still safe on real hardware.)
|
||||||
|
|
||||||
## In-kernel test cases
|
## In-kernel test cases
|
||||||
|
|
||||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||||
|
|||||||
+117
@@ -0,0 +1,117 @@
|
|||||||
|
# Timers and time
|
||||||
|
|
||||||
|
Two different needs hide under the word "timer", and danos keeps them apart:
|
||||||
|
|
||||||
|
- **Reading the clock** — *what time is it?* A read of a free-running counter.
|
||||||
|
- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.*
|
||||||
|
|
||||||
|
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||||
|
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||||
|
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||||
|
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||||
|
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||||
|
service.**
|
||||||
|
|
||||||
|
## Why time is a syscall, not a service
|
||||||
|
|
||||||
|
The tempting microkernel move is to put a timer *driver* in user space and have
|
||||||
|
applications ask it for the time over IPC. For a **monotonic clock that is wrong** —
|
||||||
|
reading `now()` should never cost an IPC round trip. The kernel is already holding the
|
||||||
|
answer: it computes the current time every time it schedules, from the TSC, in a couple
|
||||||
|
of instructions. Surfacing that as a system call is pure mechanism; routing it through a
|
||||||
|
message to another process would be slower *and* redundant, and a device like the HPET
|
||||||
|
(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`.
|
||||||
|
|
||||||
|
This is the same conclusion every serious system reaches: Linux and Zircon read the
|
||||||
|
counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the
|
||||||
|
cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall.
|
||||||
|
|
||||||
|
That "from the TSC" hides a portability question, because the TSC is only a valid clock
|
||||||
|
when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*.
|
||||||
|
danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and
|
||||||
|
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||||
|
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||||
|
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||||
|
[device-interrupts.md](device-interrupts.md).
|
||||||
|
|
||||||
|
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||||
|
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||||
|
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||||
|
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||||
|
at the end; it is deliberately not built yet.
|
||||||
|
|
||||||
|
## The three system calls
|
||||||
|
|
||||||
|
Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)):
|
||||||
|
|
||||||
|
- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not
|
||||||
|
wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a
|
||||||
|
128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of
|
||||||
|
resolution, and just an `rdtsc` plus a multiply.
|
||||||
|
- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake
|
||||||
|
deadline and the tick sweep wakes it (`scheduler.sleep`).
|
||||||
|
- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a
|
||||||
|
**timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a
|
||||||
|
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||||
|
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||||
|
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||||
|
[device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||||
|
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||||
|
both riding the scheduler tick.
|
||||||
|
|
||||||
|
## `runtime.time` — the generic interface
|
||||||
|
|
||||||
|
Applications don't call the syscalls directly; they use `runtime.time`
|
||||||
|
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||||
|
front door, not new mechanism.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const time = @import("runtime").time;
|
||||||
|
|
||||||
|
const start = time.now(); // Instant — monotonic
|
||||||
|
doWork();
|
||||||
|
const took = start.elapsed(); // Duration
|
||||||
|
time.sleep(time.Duration.fromMillis(5)); // block ~5 ms
|
||||||
|
|
||||||
|
// A deadline delivered as a notification, so a service keeps serving meanwhile:
|
||||||
|
_ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||||
|
```
|
||||||
|
|
||||||
|
- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/
|
||||||
|
fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's
|
||||||
|
millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and
|
||||||
|
returns early. All arithmetic saturates rather than wraps.
|
||||||
|
- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` —
|
||||||
|
built for deadline loops (`while (!deadline.reached()) …`).
|
||||||
|
- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is
|
||||||
|
calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller
|
||||||
|
that needs real time can treat 0 as "unavailable" rather than assume it advances).
|
||||||
|
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||||
|
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||||
|
|
||||||
|
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||||
|
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||||
|
|
||||||
|
## Wall-clock time (not built)
|
||||||
|
|
||||||
|
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||||
|
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||||
|
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||||
|
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||||
|
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||||
|
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||||
|
owns covers every current use.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||||
|
|
||||||
|
```
|
||||||
|
$ zig build test # includes library/runtime/time.zig
|
||||||
|
```
|
||||||
|
|
||||||
|
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||||
|
`Duration`, read `now()` again, and the second reading is later — the kernel's timer
|
||||||
|
driving a ring-3 program with no service in between.
|
||||||
@@ -0,0 +1,349 @@
|
|||||||
|
# Running Zig on danos: the self-hosting roadmap
|
||||||
|
|
||||||
|
A design note (not built yet) on the path to making danos a **real Zig target** — a
|
||||||
|
target you can name (`-target x86_64-danos`) and, eventually, run the Zig compiler
|
||||||
|
itself on. It is forward-looking, like [vision.md](vision.md): it sets a direction
|
||||||
|
and the decisions that follow from it, so the code we write now bends toward it
|
||||||
|
instead of away.
|
||||||
|
|
||||||
|
This note deliberately does **not** cover a text editor or terminal. Those are
|
||||||
|
easier (single-process, I/O-bound) and fall out of the early phases here almost for
|
||||||
|
free; the hard, shaping problem is the standard-library surface, so that is what
|
||||||
|
this roadmap is about.
|
||||||
|
|
||||||
|
The analysis behind it was done against **Zig 0.16** (the pinned toolchain). Zig's
|
||||||
|
standard library moves between releases — especially the parts described here — so
|
||||||
|
treat upstream references as "the shape in 0.16.x," and expect to re-check them on a
|
||||||
|
toolchain bump.
|
||||||
|
|
||||||
|
## The win condition
|
||||||
|
|
||||||
|
danos runs the Zig compiler when a bare
|
||||||
|
|
||||||
|
```
|
||||||
|
zig build-exe hello.zig
|
||||||
|
```
|
||||||
|
|
||||||
|
completes **on danos** and produces a runnable danos binary. Note the milestone is
|
||||||
|
`build-exe`, not `zig build`: the `zig build` runner spawns child processes (the
|
||||||
|
build steps), which needs a whole process-control surface danos does not have yet.
|
||||||
|
A single `build-exe` needs none of that (see Phase 3). Reaching `build-exe` is
|
||||||
|
"self-hosting"; reaching `zig build` is a later, separate lift.
|
||||||
|
|
||||||
|
### Non-goals
|
||||||
|
|
||||||
|
- **No Linux syscall/ABI emulation.** danos will not implement the Linux `syscall`
|
||||||
|
interface so that stock `x86_64-linux` binaries run. That is a permanent
|
||||||
|
compatibility treadmill and it inverts the microkernel design — explicitly out.
|
||||||
|
- **No musl port yet.** A musl libc port is a reasonable *later* effort (it unlocks
|
||||||
|
the C ecosystem), but it is not on the critical path to Zig-on-danos, and it is
|
||||||
|
deferred. The roadmap below is arranged so the work still pays off if musl ever
|
||||||
|
happens (see "The same surface, twice").
|
||||||
|
- **Editor/terminal are out of scope for this note** (they are downstream of Phase 1).
|
||||||
|
|
||||||
|
**On FFI.** Foreign-function interop splits the same way as the doors below. Zig-level
|
||||||
|
and C-ABI-*exposing* FFI (`extern`, `callconv(.c)`, C-ABI structs) work on a real target
|
||||||
|
immediately — and the `std.os.danos` seam is C-ABI-shaped by construction, so it is
|
||||||
|
FFI-friendly from the start. *Consuming* C libraries (`@cImport`, linking archives) is
|
||||||
|
the part that needs a libc + headers, i.e. the deferred musl door. So an eventual FFI
|
||||||
|
need reinforces keeping that door open; it does not change the plan.
|
||||||
|
|
||||||
|
## The realization that shapes everything: 0.16 gives us *one* seam
|
||||||
|
|
||||||
|
The instinct "to target Zig we'd have to reimplement all the `std` namespaces" was
|
||||||
|
how older Zig worked. Zig 0.16 (post-"writergate") is far kinder:
|
||||||
|
|
||||||
|
- **`std.fs` is essentially gone.** It is now path helpers plus deprecated aliases;
|
||||||
|
there is no `std.fs.File`, `std.fs.Dir`, or `std.fs.cwd()`. File and directory
|
||||||
|
work goes through **`std.Io`** — a single runtime **vtable** (`Io.zig`) of
|
||||||
|
function pointers handed to `main` as `std.process.Init.io`. `std.Io.File` and
|
||||||
|
`std.Io.Dir` are thin forwarders to that vtable. `Io.zig` and the `fs` shim carry
|
||||||
|
**zero** per-OS branches.
|
||||||
|
- **`std.posix` is one generic body** parameterised over a single `system` module.
|
||||||
|
With no libc, `system` resolves **per target OS**: `.linux => std.os.linux`,
|
||||||
|
`.plan9 => std.os.plan9`, and so on. The generic `std.posix.read`/`write`/`open`
|
||||||
|
bodies are just `system.read(...)` plus an errno switch — *identical for every
|
||||||
|
OS*. The only variable is what `system` binds to.
|
||||||
|
- **`std.os.<tag>`** (e.g. `std/os/linux.zig`) is therefore the real porting seam: a
|
||||||
|
low-level, C-ABI-shaped module of `read/write/open/close/lseek/mmap/clock/exit/…`
|
||||||
|
plus an `errno` enum and the constant tables (`O_*`, `CLOCK_*`, `S_*`).
|
||||||
|
|
||||||
|
Put together: **to port danos we write `std.os.danos` once** — the ~30-operation
|
||||||
|
seam — and the whole `std.posix` / `std.fs` / `std.Io` tower above it lights up
|
||||||
|
generically, because none of it branches on the OS. That is a dramatically smaller
|
||||||
|
and more contained target than "reimplement the namespaces."
|
||||||
|
|
||||||
|
## Three doors, and why we take the first
|
||||||
|
|
||||||
|
| Door | What it is | Verdict |
|
||||||
|
|------|-----------|---------|
|
||||||
|
| **1. Implement the std seam** (`std.os.danos`) | Write the ~30-op `system` module over danos's native ABI + VFS; the generic std tower lights up. | **Take this.** The only door that touches neither C nor the Linux ABI. |
|
||||||
|
| **2. Port musl** | Port musl libc to danos, link Zig against it. | Defer. Good later for the *C* ecosystem; barely helps *Zig* (std only uses libc on the libc-linked path). |
|
||||||
|
| **3. Emulate the Linux ABI** | Implement Linux syscalls so stock linux binaries run. | Reject. Bottomless compatibility treadmill; against the design. |
|
||||||
|
|
||||||
|
### The same surface, twice
|
||||||
|
|
||||||
|
Doors 1 and 2 are the **same native surface at different layers**. `std.posix.read`
|
||||||
|
is `system.read(...)` + an errno switch *regardless of OS* — the only question is
|
||||||
|
whether `system` is **`std.os.danos` (Zig)** or **musl (C)**. Either way, the set of
|
||||||
|
danos-facing operations you must implement is the *same* ~30 ops, all bottoming out
|
||||||
|
in danos's native syscalls + the VFS/FAT server.
|
||||||
|
|
||||||
|
So the runtime work below is **not throwaway** if musl ever happens: you are building
|
||||||
|
the danos-native implementations of that surface either way. Door 1 just packages
|
||||||
|
them as Zig; a future musl re-uses the identical kernel/VFS operations underneath. The
|
||||||
|
two symmetries worth keeping in mind: doors 1 and 2 converge at the **top** (identical
|
||||||
|
POSIX surface); doors 2 and 3 converge at the **bottom** (unmodified musl needs the
|
||||||
|
Linux syscall ABI). Door 1 is the only one that avoids both C and Linux.
|
||||||
|
|
||||||
|
### A fork is table stakes — for any door
|
||||||
|
|
||||||
|
`std.Target.Os.Tag` is a **closed enum** baked into the compiler binary *and* into
|
||||||
|
the `std` linked with every program; `-target x86_64-danos` resolves through it. So
|
||||||
|
adding `danos` as a name requires patching and rebuilding the compiler — even the
|
||||||
|
musl door needs this. "Fork Zig" is therefore not an extra cost unique to door 1; it
|
||||||
|
is the price of admission for *any* real target. What door 1 adds on top is small and
|
||||||
|
localised (below).
|
||||||
|
|
||||||
|
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||||
|
|
||||||
|
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||||
|
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||||
|
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||||
|
|
||||||
|
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||||
|
(`read/write/open/close/lseek/mmap/munmap/clock/exit/…`) + an errno enum + the
|
||||||
|
constant tables, each backed by danos's native syscalls and the VFS. **Structure it
|
||||||
|
to mirror `std/os/linux.zig`.** This is the load-bearing, *non-throwaway* artifact:
|
||||||
|
when we fork Zig, `runtime.os` is copy-pasted (near-verbatim) into `std.os.danos`.
|
||||||
|
- **`runtime.fs` — the thin native file API** danos programs use *today*, layered
|
||||||
|
over `runtime.os`. It is also the concrete backing for the `std.Io` vtable's
|
||||||
|
file-write entry once we're a real target, which is why program stdout, diagnostics,
|
||||||
|
and file writes should all be *decided once at that seam* rather than as bespoke
|
||||||
|
per-call helpers (see "How this informs decisions now").
|
||||||
|
|
||||||
|
**Do not hand-mirror the high-level std namespaces.** `std.fs`/`std.Io`/`std.process`
|
||||||
|
are generic and OS-agnostic; once `std.os.danos` exists and we fork, upstream *gives*
|
||||||
|
them to danos for free. Hand-writing `runtime.std.fs` to imitate them would be
|
||||||
|
redundant the day the fork works, and it would chase a moving target (0.16's `std.Io`
|
||||||
|
is large and still shifting). Build the seam well; take the tower for free.
|
||||||
|
|
||||||
|
**Why not a library called `std`?** Because `@import("std")` resolves to the
|
||||||
|
compiler-provided standard library; a user module named `std` would *shadow* it for
|
||||||
|
anything that imports it that way. That is the real reason the seam lives *inside* a
|
||||||
|
forked std as `std/os/danos.zig`, not as a `runtime.std` library — and why danos's end
|
||||||
|
state (`@import("std")` just working, and knowing danos) is the most natively Zig it can
|
||||||
|
be. `runtime.os` is only the interim staging ground: developed against the stock
|
||||||
|
toolchain so Phase 1 need not wait on the fork, then promoted near-verbatim into the
|
||||||
|
fork's `std/os/danos.zig`.
|
||||||
|
|
||||||
|
### Retire `library/posix`
|
||||||
|
|
||||||
|
The `posix` compatibility layer (`unistd`, `stdio`) was the right instinct too early.
|
||||||
|
Its whole value is POSIX *spellings* for POSIX software — and danos has no POSIX
|
||||||
|
software; every current caller is danos-native code that could use `runtime.fs`
|
||||||
|
directly. The real POSIX story arrives later and from elsewhere (musl, or upstream
|
||||||
|
`std`'s own posix over `std.os.danos`), which supersedes a hand-rolled shim. So it is
|
||||||
|
premature abstraction that adds a "which layer do I use?" fork with no payoff yet.
|
||||||
|
|
||||||
|
Its footprint is tiny: **five** call sites, all `unistd` file operations —
|
||||||
|
`system/services/fat/fat.zig` (`mount`), the `vfs-test` and `fat-test` clients, and
|
||||||
|
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` is dead — nothing
|
||||||
|
imports it. The plan: build `runtime.fs`, migrate those five to it, delete
|
||||||
|
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary`.
|
||||||
|
|
||||||
|
## Where danos stands: coverage vs. the gaps
|
||||||
|
|
||||||
|
What the seam needs, and what danos already provides:
|
||||||
|
|
||||||
|
| std need | danos today | Gap |
|
||||||
|
|----------|-------------|-----|
|
||||||
|
| open / read / write / close / lseek | VFS (via the current `unistd`, → `runtime.fs`) | none — repackage |
|
||||||
|
| directory read (`getdents`) | VFS `readdir` | none — repackage |
|
||||||
|
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||||
|
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||||
|
| monotonic clock | `clock` syscall | none |
|
||||||
|
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||||
|
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||||
|
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||||
|
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||||
|
| wall-clock / realtime | done — `wall_clock` syscall (CMOS RTC, Phase 2d) | — |
|
||||||
|
| **environment variables** | `Init` has no env field | missing (can start empty) |
|
||||||
|
| **cwd / chdir** | paths are absolute or bare | missing (no cwd anchor) |
|
||||||
|
| **entropy / random** | — | missing (needed behind `vtable.random`) |
|
||||||
|
| process spawn + exit status | `system_spawn` starts a *named ramdisk binary*; `ExitReason` is a *category* | no exec-of-path, no numeric `WEXITSTATUS` |
|
||||||
|
| threads | one thread per process | avoided via `-fsingle-threaded` (below) |
|
||||||
|
| symlinks | `NodeKind` has the tag; unimplemented | low priority |
|
||||||
|
|
||||||
|
The clustering is clear: reads and memory are basically done; the real work is
|
||||||
|
**filesystem mutation + richer stat + wall-clock**, and a few small seam pieces
|
||||||
|
(page-allocator hook, stdio bytes, entropy). Process spawning and threads are
|
||||||
|
side-stepped entirely for a single `build-exe`.
|
||||||
|
|
||||||
|
## The roadmap
|
||||||
|
|
||||||
|
### Phase 0 — Make `danos` a real target
|
||||||
|
|
||||||
|
**Host, target, self-host — keep the three roles straight.** The *host* is where the
|
||||||
|
compiler runs (your mac + linux dev machines); the *target* is what it emits (`danos`);
|
||||||
|
and eventually danos becomes a host too (self-hosting — the win condition). So the move
|
||||||
|
is: fork the compiler, build it **for** your dev hosts, and teach it to **cross-compile
|
||||||
|
to** danos. You already do this — danos is cross-compiled `freestanding` from your dev
|
||||||
|
host today; Phase 0 swaps that `freestanding` target for a real `x86_64-danos` one, which
|
||||||
|
is what unlocks the native `std`.
|
||||||
|
|
||||||
|
**Why a compiler fork, not just a `--zig-lib-dir` override.** `std.Target.Os.Tag` is a
|
||||||
|
*closed enum compiled into the compiler binary*, so `-target x86_64-danos` will not even
|
||||||
|
parse unless the compiler itself knows the tag. Overriding the std lib directory alone
|
||||||
|
cannot add a target — and there is no libc-only shortcut (a future musl needs the same
|
||||||
|
patch). The only alternative, staying on `freestanding` + hand-shims, is exactly the
|
||||||
|
non-native feel we are leaving: `@import("std")` there is stubbed, not real.
|
||||||
|
|
||||||
|
**The fork.** Clone `ziglang/zig` at the pinned 0.16 tag; build it with a stock
|
||||||
|
same-version `zig` (`zig build` in the tree — a standard, LLVM-pulling, roughly one-time
|
||||||
|
build); point danos's `build.zig`/CI at the resulting binary. Four localised patches:
|
||||||
|
|
||||||
|
- add `danos` to `std.Target.Os.Tag`, in the "no version range" group alongside
|
||||||
|
plan9/serenity;
|
||||||
|
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||||
|
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||||
|
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||||
|
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||||
|
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||||
|
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||||
|
not a prerequisite for starting).
|
||||||
|
|
||||||
|
This is the fork treadmill we accept once. Keep the patch set tiny and `else`-friendly,
|
||||||
|
pin to one 0.16.x, and rebase on point releases.
|
||||||
|
|
||||||
|
### Phase 1 — `runtime.os` read-side + allocator + stdio + cwd; retire `posix`
|
||||||
|
|
||||||
|
Author `runtime.os` (→ `std.os.danos`): the `errno` enum, the constant tables, and
|
||||||
|
the C-convention `read / write / open / openat / close / lseek / mmap / munmap /
|
||||||
|
exit`, each returning result-or-`-errno`. Most backing already exists (VFS + native
|
||||||
|
mmap + clock).
|
||||||
|
|
||||||
|
- Provide `page_allocator` via `root.os.heap.page_allocator` (a thin override over
|
||||||
|
danos `mmap`). This sits **outside** the `std.Io` vtable, so it is wired separately.
|
||||||
|
- Wire fd 0/1/2 to a console **byte** stream (today output only reaches `debug_write`;
|
||||||
|
input is structured `InputEvent` IPC — a byte tty is a new, small thing in both
|
||||||
|
directions).
|
||||||
|
- Add a `getcwd`/`chdir` anchor so `std.fs.cwd()`-style resolution has something to
|
||||||
|
resolve against.
|
||||||
|
- Build `runtime.fs` over `runtime.os`; migrate the five `posix` callers to it; delete
|
||||||
|
`library/posix/` and drop its build module.
|
||||||
|
|
||||||
|
After Phase 1, the surface an editor or terminal needs (open/read/write/close/lseek/
|
||||||
|
readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||||
|
|
||||||
|
### Phase 2 — Filesystem mutation + real stat (the compiler's cache tower)
|
||||||
|
|
||||||
|
danos's biggest genuine gap, and the correctness-critical one:
|
||||||
|
|
||||||
|
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||||
|
([protocol.zig](../system/services/vfs/protocol.zig)) and the FAT engine
|
||||||
|
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||||
|
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||||
|
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||||
|
(danos is monotonic-only today; an RTC/time service is the dependency).
|
||||||
|
|
||||||
|
Because `std.fs`/`std.Io` have no per-OS branches, finishing this in `runtime.os`
|
||||||
|
lights up the whole file tower for the compiler at once. Environment can stay an empty
|
||||||
|
map until the kernel populates a non-empty `envp`.
|
||||||
|
|
||||||
|
**Status — Phase 2 complete.** `truncate` (O_TRUNC, closing the boot-log stale-tail
|
||||||
|
bug), `mkdir`, `unlink`, and `rename` are all wired through the FAT engine, the VFS
|
||||||
|
protocol + router, and `runtime.fs` (`makeDirectory` / `remove` / `rename`) —
|
||||||
|
host-tested and QEMU-tested (`fat-mutations` + `fat-rename` make a directory, write+read
|
||||||
|
a file in it, rename it, then remove it through the mount). `removeFile` and `rename`
|
||||||
|
are LFN-aware; `rename` is same-directory + 8.3 (cross-directory and long-name-
|
||||||
|
preserving rename are noted limitations). Wall-clock is now a kernel syscall
|
||||||
|
(`wall_clock`, a CMOS-RTC read anchored to the monotonic clock), and the FAT engine
|
||||||
|
stamps and reports **mtime** — `stat` / `runtime.fs.Attributes` carry a real
|
||||||
|
modification time (the `fat-mtime` case reads it back within seconds of the host clock).
|
||||||
|
The remaining `stat` fields, `mode`/`inode`, are deferred (not needed until the
|
||||||
|
compiler's cache layer wants them). **Everything past here is gated on Phase 0 (the
|
||||||
|
fork):** the `runtime.os` seam, `cwd`, stdio-as-fds, and the compiler bring-up.
|
||||||
|
|
||||||
|
### Phase 3 — Single-threaded, self-linked compiler bring-up
|
||||||
|
|
||||||
|
Build the compiler with **two load-bearing flags**:
|
||||||
|
|
||||||
|
- **`-fsingle-threaded`** removes `std.Thread` entirely — `Thread.spawn` is a hard
|
||||||
|
compile error under it, and `std.Io`'s threaded backend runs inline. danos being
|
||||||
|
one-thread-per-process is therefore **not** a blocker. Parallel codegen is a
|
||||||
|
throughput optimisation, not a correctness requirement.
|
||||||
|
- **`-fno-llvm -fno-lld`** keeps codegen and linking **in-process** (the self-hosted
|
||||||
|
x86-64 backend + self-linker), so a single `build-exe` **never forks a child**. That
|
||||||
|
is what lets us defer the entire spawn/exec/wait surface.
|
||||||
|
|
||||||
|
Then supply the few remaining seam pieces: `now` (wrap the danos clock), an entropy
|
||||||
|
source behind `vtable.random` (`randomSecure` can alias it initially — low volume, for
|
||||||
|
temp-file names and hashmap seeds), and the Phase-2 mkdir/rename/unlink for cache dir
|
||||||
|
trees and atomic temp-then-rename output.
|
||||||
|
|
||||||
|
**Explicitly deferred** (not on the `build-exe` path): child-process spawn/exec (only
|
||||||
|
`zig build` and external tools need it), `std.Thread`, `fsync` (FAT is write-through
|
||||||
|
today), symlinks, and musl.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **The std-fork rebase treadmill is the main ongoing cost.** A new OS tag touches the
|
||||||
|
same broad file set plan9/serenity touch (hundreds of `native_os` sites, plus
|
||||||
|
"unsupported OS" `@compileError` dead-ends a new tag must be routed around), and the
|
||||||
|
entire `std.Io` layer is new in 0.16 and still moving. Stay pinned to one 0.16.x,
|
||||||
|
keep additions localised and `else`-friendly. Watch the closed-enum gotcha: adding
|
||||||
|
`danos` to `Os.Tag` can break existing *exhaustive* switches that lack an `else`, so
|
||||||
|
expect to touch switch sites beyond the ones you implement.
|
||||||
|
- **Single-threaded is load-bearing.** The "no `std.Thread`" simplification rests
|
||||||
|
entirely on `-fsingle-threaded`. If a dependency or flag flips threading back on, you
|
||||||
|
inherit an unescapable compile error (no root-hook exists) — the only outs are a full
|
||||||
|
thread-impl fork or linking libc for pthreads. Keep `single_threaded` asserted end to
|
||||||
|
end.
|
||||||
|
- **In-process linking is load-bearing.** Reaching the compiler without fork/exec
|
||||||
|
depends on `-fno-llvm -fno-lld`. The moment you shell out to LLD/`ld`, you need the
|
||||||
|
full `spawn`/`wait` surface — the hardest microkernel piece — and danos's
|
||||||
|
`system_spawn` only starts a *named ramdisk binary*, not exec of an arbitrary path.
|
||||||
|
Verify the self-hosted backend covers the target output before assuming child
|
||||||
|
processes are optional.
|
||||||
|
- **The shim cannot host the compiler.** danos's current `runtime`/`posix` is fine for
|
||||||
|
danos's *own* native programs, but the compiler `import`s *upstream* `std`, which on
|
||||||
|
a non-target hits the void `system` stub. So the compiler forces the real target
|
||||||
|
(Phase 0's fork). Do not over-invest in extending the hand-shim for compiler
|
||||||
|
purposes; put that effort into `runtime.os` + the VFS/FAT operations, which both the
|
||||||
|
fork *and* a future musl consume.
|
||||||
|
- **`"w"`/`O_CREAT` does not truncate — a silent-corruption bug on this road.** The FAT
|
||||||
|
engine's `writeFile` only *grows* `node.size`, so overwriting a shorter file leaves
|
||||||
|
trailing garbage. Harmless for the boot log today, but for a compiler it means
|
||||||
|
**corrupt `.o`/cache files that look like nondeterministic compiler bugs.** Land
|
||||||
|
`truncate` (Phase 2) before the compiler ever writes cache.
|
||||||
|
- **Exit status is categorical, not numeric.** `process_exit_reason` returns an
|
||||||
|
`ExitReason` *category*, not a numeric code (`WEXITSTATUS`). Fine while spawn is
|
||||||
|
stubbed; the day `zig build` or external tools arrive, plan a kernel exit-record
|
||||||
|
extension — do not let it surprise you.
|
||||||
|
|
||||||
|
## How this informs decisions now
|
||||||
|
|
||||||
|
Two current decisions fall out of this roadmap:
|
||||||
|
|
||||||
|
1. **The `runtime.fs` / `std.Io` question resolves at the vtable seam.** Because 0.16
|
||||||
|
routes *all* output through the `std.Io` vtable's file-write entry, and stdout/stderr
|
||||||
|
are just `File`s with well-known handles, build `runtime.fs` (and the console stdout)
|
||||||
|
as the concrete backing for that entry — not as a bespoke `std.Io.Writer`-only shim.
|
||||||
|
Decide it once, at the seam, and program stdout, diagnostics, and file writes all
|
||||||
|
flow through the same danos VFS/console path.
|
||||||
|
2. **The boot-log `truncate` caveat is now fixed** (Phase 2a). It was the same
|
||||||
|
`writeFile`-only-grows gap that on the self-hosting road would corrupt build output;
|
||||||
|
`engine.truncate` + an O_TRUNC open flag now free the old chain so a shorter rewrite
|
||||||
|
leaves no stale tail, and the boot-log flush opens with it.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [vision.md](vision.md) — the north star this serves.
|
||||||
|
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||||
|
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||||
|
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||||
|
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||||
|
filesystem layout the file surface serves.
|
||||||
|
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||||
|
are confined, and now retired).
|
||||||
@@ -1,13 +0,0 @@
|
|||||||
//! DanOS's POSIX / C compatibility layer — `unistd`, `stdio`, and (later) the C
|
|
||||||
//! `errno` / `struct stat` / `extern "C"` surface. This is the *one* place POSIX and
|
|
||||||
//! C spellings are allowed to appear verbatim (see docs/coding-standards.md): a file
|
|
||||||
//! under library/posix/ *is* the foreign ABI, so it keeps the ABI's names. Everything
|
|
||||||
//! it touches on the danos side (the VFS protocol, the runtime) uses danos names,
|
|
||||||
//! which this layer translates to at the boundary.
|
|
||||||
//!
|
|
||||||
//! It is layered strictly *over* the runtime: it calls the runtime's IPC and heap,
|
|
||||||
//! never the kernel's system calls directly. danos-native applications use the
|
|
||||||
//! runtime; this exists so *POSIX* software can too.
|
|
||||||
|
|
||||||
pub const unistd = @import("unistd.zig");
|
|
||||||
pub const stdio = @import("stdio.zig");
|
|
||||||
@@ -1,115 +0,0 @@
|
|||||||
//! A small C stdio layer over the POSIX-style file API (unistd.zig). Unbuffered
|
|
||||||
//! for now — each fread/fwrite is one VFS round trip; an internal buffer (fewer
|
|
||||||
//! IPC calls) is a later optimisation. Both a Zig-callable API and `extern "C"`
|
|
||||||
//! symbols are provided, so Zig and future C programs share it.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const unistd = @import("unistd.zig");
|
|
||||||
const heap = @import("runtime").heap;
|
|
||||||
|
|
||||||
pub const SEEK_SET = unistd.SEEK_SET;
|
|
||||||
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
|
||||||
pub const SEEK_END = unistd.SEEK_END;
|
|
||||||
|
|
||||||
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
|
||||||
/// heap; `fclose` frees it.
|
|
||||||
pub const FILE = extern struct {
|
|
||||||
fd: i32,
|
|
||||||
eof: c_int = 0,
|
|
||||||
err: c_int = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
fn flagsFor(mode: []const u8) u32 {
|
|
||||||
if (mode.len == 0) return 0;
|
|
||||||
return switch (mode[0]) {
|
|
||||||
'w', 'a' => unistd.O_CREAT,
|
|
||||||
else => 0,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Open `path` in `mode` ("r"/"w"/"a", '+' ignored for now). Returns null on error.
|
|
||||||
pub fn fopen(path: []const u8, mode: []const u8) ?*FILE {
|
|
||||||
const fd = unistd.open(path, flagsFor(mode));
|
|
||||||
if (fd < 0) return null;
|
|
||||||
const f = heap.allocator().create(FILE) catch {
|
|
||||||
unistd.close(fd);
|
|
||||||
return null;
|
|
||||||
};
|
|
||||||
f.* = .{ .fd = fd };
|
|
||||||
if (mode.len > 0 and mode[0] == 'a') _ = unistd.lseek(fd, 0, unistd.SEEK_END);
|
|
||||||
return f;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn fclose(f: *FILE) c_int {
|
|
||||||
unistd.close(f.fd);
|
|
||||||
heap.allocator().destroy(f);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
|
||||||
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
|
||||||
const total = size * nmemb;
|
|
||||||
if (total == 0) return 0;
|
|
||||||
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
|
||||||
if (n <= 0) {
|
|
||||||
f.eof = 1;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
return @as(usize, @intCast(n)) / size;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write `size*nmemb` bytes; returns the number of whole items written.
|
|
||||||
pub fn fwrite(data: []const u8, size: usize, nmemb: usize, f: *FILE) usize {
|
|
||||||
const total = @min(data.len, size * nmemb);
|
|
||||||
if (total == 0) return 0;
|
|
||||||
const n = unistd.write(f.fd, data[0..total]);
|
|
||||||
if (n <= 0) {
|
|
||||||
f.err = 1;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
return @as(usize, @intCast(n)) / size;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
|
||||||
f.eof = 0;
|
|
||||||
return if (unistd.lseek(f.fd, off, whence) < 0) -1 else 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn ftell(f: *FILE) i64 {
|
|
||||||
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn rewind(f: *FILE) void {
|
|
||||||
_ = fseek(f, 0, SEEK_SET);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn feof(f: *FILE) c_int {
|
|
||||||
return f.eof;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn ferror(f: *FILE) c_int {
|
|
||||||
return f.err;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn fputs(s: []const u8, f: *FILE) c_int {
|
|
||||||
return if (unistd.write(f.fd, s) < 0) -1 else 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn fputc(c: u8, f: *FILE) c_int {
|
|
||||||
const b = [_]u8{c};
|
|
||||||
return if (unistd.write(f.fd, &b) == 1) c else -1;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn fgetc(f: *FILE) c_int {
|
|
||||||
var b: [1]u8 = undefined;
|
|
||||||
const n = unistd.read(f.fd, &b);
|
|
||||||
if (n <= 0) {
|
|
||||||
f.eof = 1;
|
|
||||||
return -1; // EOF
|
|
||||||
}
|
|
||||||
return b[0];
|
|
||||||
}
|
|
||||||
|
|
||||||
// Real `extern "C"` symbols (fopen/fread/fseek/...) — with a C-string signature
|
|
||||||
// distinct from the Zig slice API above — land with the first C program, wired
|
|
||||||
// via @export so they don't collide with these Zig names.
|
|
||||||
@@ -1,148 +0,0 @@
|
|||||||
//! POSIX-style file API for user programs — the low level under C stdio. Files
|
|
||||||
//! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
|
|
||||||
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
|
|
||||||
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const protocol = @import("vfs-protocol");
|
|
||||||
const ipc = @import("runtime").ipc;
|
|
||||||
|
|
||||||
pub const O_CREAT = protocol.create;
|
|
||||||
pub const SEEK_SET: u32 = 0;
|
|
||||||
pub const SEEK_CURRENT: u32 = 1;
|
|
||||||
pub const SEEK_END: u32 = 2;
|
|
||||||
|
|
||||||
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
|
||||||
var vfs_handle: usize = 0;
|
|
||||||
var vfs_resolved = false;
|
|
||||||
fn vfs() ?usize {
|
|
||||||
if (!vfs_resolved) {
|
|
||||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
|
||||||
vfs_resolved = true;
|
|
||||||
}
|
|
||||||
return vfs_handle;
|
|
||||||
}
|
|
||||||
|
|
||||||
const maximum_fds = 32;
|
|
||||||
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
|
||||||
var fds = [_]Fd{.{}} ** maximum_fds;
|
|
||||||
|
|
||||||
fn allocFd() ?usize {
|
|
||||||
for (&fds, 0..) |*f, i| {
|
|
||||||
if (!f.used) {
|
|
||||||
f.* = .{ .used = true };
|
|
||||||
return i;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
|
||||||
|
|
||||||
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
|
||||||
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
|
||||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
|
||||||
const h = vfs() orelse return null;
|
|
||||||
var message: [protocol.message_maximum]u8 = undefined;
|
|
||||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
|
||||||
const slen = @min(send.len, protocol.maximum_payload);
|
|
||||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
|
||||||
|
|
||||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
|
||||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
|
||||||
if (n < protocol.reply_size) return null;
|
|
||||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
|
||||||
const rpl = @min(n - protocol.reply_size, out.len);
|
|
||||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
|
||||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
|
||||||
pub fn open(path: []const u8, flags: u32) i32 {
|
|
||||||
const fd = allocFd() orelse return -1;
|
|
||||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
|
||||||
const r = transact(request, path, &.{}) orelse {
|
|
||||||
fds[fd].used = false;
|
|
||||||
return -1;
|
|
||||||
};
|
|
||||||
if (r.reply.status != 0) {
|
|
||||||
fds[fd].used = false;
|
|
||||||
return -1;
|
|
||||||
}
|
|
||||||
fds[fd] = .{ .used = true, .node = r.reply.node, .offset = 0 };
|
|
||||||
return @intCast(fd);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn fdPtr(fd: i32) ?*Fd {
|
|
||||||
if (fd < 0 or fd >= maximum_fds) return null;
|
|
||||||
const f = &fds[@intCast(fd)];
|
|
||||||
return if (f.used) f else null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
|
||||||
pub fn read(fd: i32, buffer: []u8) isize {
|
|
||||||
const f = fdPtr(fd) orelse return -1;
|
|
||||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
|
||||||
const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
|
||||||
const r = transact(request, &.{}, buffer) orelse return -1;
|
|
||||||
if (r.reply.status != 0) return -1;
|
|
||||||
f.offset += r.reply.len;
|
|
||||||
return @intCast(r.reply.len);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write `data` at the current offset; returns the count or -1.
|
|
||||||
pub fn write(fd: i32, data: []const u8) isize {
|
|
||||||
const f = fdPtr(fd) orelse return -1;
|
|
||||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
|
||||||
const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
|
||||||
const r = transact(request, data[0..want], &.{}) orelse return -1;
|
|
||||||
if (r.reply.status != 0) return -1;
|
|
||||||
f.offset += r.reply.len;
|
|
||||||
return @intCast(r.reply.len);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Reposition the fd's offset. Returns the new offset or -1. (SEEK_END needs the
|
|
||||||
/// file size, which `stat` provides; handled by fetching it here.)
|
|
||||||
pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
|
||||||
const f = fdPtr(fd) orelse return -1;
|
|
||||||
const base: i64 = switch (whence) {
|
|
||||||
SEEK_SET => 0,
|
|
||||||
SEEK_CURRENT => @intCast(f.offset),
|
|
||||||
SEEK_END => blk: {
|
|
||||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
|
||||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
|
||||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
|
||||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
|
||||||
const st = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
|
||||||
break :blk @intCast(st.size);
|
|
||||||
},
|
|
||||||
else => return -1,
|
|
||||||
};
|
|
||||||
const pos = base + off;
|
|
||||||
if (pos < 0) return -1;
|
|
||||||
f.offset = @intCast(pos);
|
|
||||||
return pos;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Stat `path`. Returns 0 or -1.
|
|
||||||
pub fn stat(path: []const u8, out: *protocol.FileStatus) i32 {
|
|
||||||
// Open, stat by node, close — simple and enough for now.
|
|
||||||
const fd = open(path, 0);
|
|
||||||
if (fd < 0) return -1;
|
|
||||||
defer close(fd);
|
|
||||||
const f = fdPtr(fd).?;
|
|
||||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
|
||||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
|
||||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
|
||||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
|
||||||
out.* = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Close an fd (best effort — tells the VFS to release the open file).
|
|
||||||
pub fn close(fd: i32) void {
|
|
||||||
const f = fdPtr(fd) orelse return;
|
|
||||||
const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
|
||||||
_ = transact(request, &.{}, &.{});
|
|
||||||
f.used = false;
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
//! Block-device client: the helper a filesystem uses to read and write a block
|
||||||
|
//! device (a USB stick, via usb-storage) without hand-rolling the block-protocol
|
||||||
|
//! IPC. Layered over `ipc` and the shared `block-protocol` wire format, like
|
||||||
|
//! `runtime.usb` over the transfer protocol.
|
||||||
|
//!
|
||||||
|
//! Transfers name a caller-owned DMA buffer by physical address (from
|
||||||
|
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||||
|
//! limit — the same handoff usb-storage uses toward the controller.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("block-protocol");
|
||||||
|
|
||||||
|
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||||
|
|
||||||
|
pub const Device = struct {
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
|
||||||
|
/// The device's block size and total block count.
|
||||||
|
pub fn geometry(self: Device) ?Geometry {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||||
|
if (n < protocol.reply_size) return null;
|
||||||
|
const result = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||||
|
if (result.status != 0) return null;
|
||||||
|
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||||
|
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||||
|
return self.transfer(.read, lba, count, physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||||
|
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||||
|
return self.transfer(.write, lba, count, physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||||
|
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||||
|
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||||
|
pub fn flush(self: Device) bool {
|
||||||
|
return self.transfer(.flush, 0, 0, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (n < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Look up the block device, retrying generously while the USB storage chain
|
||||||
|
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||||
|
pub fn open() ?Device {
|
||||||
|
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||||
|
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||||
|
// tens of seconds under emulation.
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 1200) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
@@ -0,0 +1,198 @@
|
|||||||
|
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||||
|
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||||
|
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
||||||
|
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("display-protocol");
|
||||||
|
|
||||||
|
/// The display's current mode, as `info()` reports it.
|
||||||
|
pub const Info = struct {
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||||
|
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The service endpoint, looked up once and cached.
|
||||||
|
var handle: ?ipc.Handle = null;
|
||||||
|
|
||||||
|
/// Look up the display service, retrying while it comes up (a client races its
|
||||||
|
/// registration at boot). Returns the endpoint, or null if it never appears.
|
||||||
|
fn service() ?ipc.Handle {
|
||||||
|
if (handle) |h| return h;
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.display)) |h| {
|
||||||
|
handle = h;
|
||||||
|
return h;
|
||||||
|
}
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||||
|
/// so callers can read `info`/`layer` fields on success.
|
||||||
|
fn transact(request: protocol.Request, out: *protocol.Reply) bool {
|
||||||
|
const h = service() orelse return false;
|
||||||
|
var req = request;
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
out.* = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||||
|
return out.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The display's current mode, or null if the service never came up.
|
||||||
|
pub fn info() ?Info {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
if (!transact(.{ .operation = @intFromEnum(protocol.Operation.info) }, &reply)) return null;
|
||||||
|
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Composite the dirty layers and flush the frame to the screen.
|
||||||
|
pub fn present() bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One selectable display mode.
|
||||||
|
pub const Mode = protocol.Mode;
|
||||||
|
|
||||||
|
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||||
|
/// (zero on the GOP floor, or if the service never came up).
|
||||||
|
pub fn modes(out: []Mode) usize {
|
||||||
|
const h = service() orelse return 0;
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||||
|
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||||
|
if (len < protocol.modes_reply_size) return 0;
|
||||||
|
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||||
|
if (answer.status != 0) return 0;
|
||||||
|
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||||
|
for (0..count) |i| out[i] = answer.modes[i];
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||||
|
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||||
|
pub fn setMode(width: u32, height: u32) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||||
|
if (changed) mode = null; // the cached mode is stale now
|
||||||
|
return changed;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||||
|
var mode: ?Info = null;
|
||||||
|
|
||||||
|
fn cachedInfo() ?Info {
|
||||||
|
if (mode) |m| return m;
|
||||||
|
const i = info() orelse return null;
|
||||||
|
mode = i;
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||||
|
/// client packs colours through this so it never has to know the byte order itself.
|
||||||
|
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||||
|
const format = if (cachedInfo()) |i| i.format else 0;
|
||||||
|
return protocol.pack(format, r, g, b);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||||
|
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||||
|
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||||
|
pub const Layer = struct {
|
||||||
|
id: u32,
|
||||||
|
|
||||||
|
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||||
|
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.fill_rect),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
.colour = colour,
|
||||||
|
}, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||||
|
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||||
|
/// `protocol.maximum_payload`.
|
||||||
|
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||||
|
var request = protocol.Request{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.blit_tile),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
};
|
||||||
|
const header = std.mem.asBytes(&request);
|
||||||
|
if (header.len + pixels.len > protocol.message_maximum) return false;
|
||||||
|
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(buffer[0..header.len], header);
|
||||||
|
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||||
|
const h_svc = service() orelse return false;
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Move / restack / show or hide the layer.
|
||||||
|
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.configure_layer),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.z = z,
|
||||||
|
.visible = if (visible) 1 else 0,
|
||||||
|
}, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||||
|
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||||
|
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.damage),
|
||||||
|
.layer = self.id,
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
}, &reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the layer and its surface.
|
||||||
|
pub fn destroy(self: Layer) bool {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
return transact(.{ .operation = @intFromEnum(protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||||
|
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||||
|
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||||
|
var reply: protocol.Reply = undefined;
|
||||||
|
if (!transact(.{
|
||||||
|
.operation = @intFromEnum(protocol.Operation.create_layer),
|
||||||
|
.x = @bitCast(x),
|
||||||
|
.y = @bitCast(y),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
.z = z,
|
||||||
|
.visible = 1,
|
||||||
|
}, &reply)) return null;
|
||||||
|
return .{ .id = reply.layer };
|
||||||
|
}
|
||||||
@@ -0,0 +1,274 @@
|
|||||||
|
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||||
|
//! lists files served by the user-space VFS (system/services/vfs), each call
|
||||||
|
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||||
|
//! danos programs use directly; it is also where the file operations that later
|
||||||
|
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||||
|
//! the old POSIX `unistd` shim — a compatibility spelling danos does not need yet.
|
||||||
|
//!
|
||||||
|
//! Handles are *values*, not entries in a global descriptor table: a `File` /
|
||||||
|
//! `Directory` owns its VFS node id and (for files) a byte offset. So there is no
|
||||||
|
//! per-process fd limit and no shared table to synchronise — the danos-native
|
||||||
|
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
|
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||||
|
/// wire protocol.
|
||||||
|
pub const Kind = protocol.NodeKind;
|
||||||
|
|
||||||
|
/// A node's metadata (the answer to a status request).
|
||||||
|
pub const Attributes = struct {
|
||||||
|
size: u64,
|
||||||
|
kind: Kind,
|
||||||
|
/// Modification time — Unix epoch seconds, UTC. 0 if the filesystem has none.
|
||||||
|
mtime: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Map a wire `NodeKind` value to the enum, defaulting anything unrecognised to
|
||||||
|
// `.regular` (the server is trusted, but a value outside the enum would be
|
||||||
|
// illegal to `@enumFromInt` directly).
|
||||||
|
fn kindFromWire(value: u32) Kind {
|
||||||
|
return switch (value) {
|
||||||
|
@intFromEnum(Kind.directory) => .directory,
|
||||||
|
@intFromEnum(Kind.character_device) => .character_device,
|
||||||
|
@intFromEnum(Kind.block_device) => .block_device,
|
||||||
|
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||||
|
@intFromEnum(Kind.fifo) => .fifo,
|
||||||
|
@intFromEnum(Kind.socket) => .socket,
|
||||||
|
else => .regular,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How to open a path.
|
||||||
|
pub const OpenOptions = struct {
|
||||||
|
/// Create the file if it does not exist.
|
||||||
|
create: bool = false,
|
||||||
|
/// Open a directory node (for listing) rather than a file.
|
||||||
|
directory: bool = false,
|
||||||
|
/// Truncate an existing file to zero length on open (O_TRUNC) — replace its
|
||||||
|
/// contents rather than overwriting in place.
|
||||||
|
truncate: bool = false,
|
||||||
|
|
||||||
|
fn wireFlags(self: OpenOptions) u32 {
|
||||||
|
var f: u32 = 0;
|
||||||
|
if (self.create) f |= protocol.create;
|
||||||
|
if (self.directory) f |= protocol.directory;
|
||||||
|
if (self.truncate) f |= protocol.truncate;
|
||||||
|
return f;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// The VFS server endpoint, looked up once by well-known id and cached.
|
||||||
|
var vfs_handle: ipc.Handle = 0;
|
||||||
|
var vfs_resolved = false;
|
||||||
|
fn vfs() ?ipc.Handle {
|
||||||
|
if (!vfs_resolved) {
|
||||||
|
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||||
|
vfs_resolved = true;
|
||||||
|
}
|
||||||
|
return vfs_handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||||
|
|
||||||
|
// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||||
|
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||||
|
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||||
|
const h = vfs() orelse return null;
|
||||||
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||||
|
const slen = @min(send.len, protocol.maximum_payload);
|
||||||
|
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||||
|
|
||||||
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||||
|
if (n < protocol.reply_size) return null;
|
||||||
|
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||||
|
const rpl = @min(n - protocol.reply_size, out.len);
|
||||||
|
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||||
|
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||||
|
pub const File = struct {
|
||||||
|
node: u64,
|
||||||
|
offset: u64 = 0,
|
||||||
|
|
||||||
|
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||||
|
/// null on error.
|
||||||
|
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||||
|
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, &.{}, buffer) orelse return null;
|
||||||
|
if (r.reply.status != 0) return null;
|
||||||
|
self.offset += r.reply.len;
|
||||||
|
return r.reply.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `data` at the current offset; returns the count written. A single
|
||||||
|
/// call is capped at the VFS payload size, so the return may be short — use
|
||||||
|
/// `writeAll` to write the whole slice. Null on error.
|
||||||
|
pub fn write(self: *File, data: []const u8) ?usize {
|
||||||
|
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, data[0..want], &.{}) orelse return null;
|
||||||
|
if (r.reply.status != 0) return null;
|
||||||
|
self.offset += r.reply.len;
|
||||||
|
return r.reply.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||||
|
/// total written, or null if a write failed before any progress.
|
||||||
|
pub fn writeAll(self: *File, data: []const u8) ?usize {
|
||||||
|
var written: usize = 0;
|
||||||
|
while (written < data.len) {
|
||||||
|
const n = self.write(data[written..]) orelse return if (written == 0) null else written;
|
||||||
|
if (n == 0) return written; // no forward progress; stop rather than spin
|
||||||
|
written += n;
|
||||||
|
}
|
||||||
|
return written;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Move the read/write cursor to an absolute byte position.
|
||||||
|
pub fn seekTo(self: *File, position: u64) void {
|
||||||
|
self.offset = position;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This file's metadata.
|
||||||
|
pub fn attributes(self: *File) ?Attributes {
|
||||||
|
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &buffer) orelse return null;
|
||||||
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||||
|
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||||
|
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the VFS's open handle for this file.
|
||||||
|
pub fn close(self: *File) void {
|
||||||
|
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
_ = transact(request, &.{}, &.{});
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||||
|
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||||
|
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
|
||||||
|
const r = transact(request, path, &.{}) orelse return null;
|
||||||
|
if (r.reply.status != 0) return null;
|
||||||
|
return .{ .node = r.reply.node };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A path's metadata without keeping it open (open -> status -> close).
|
||||||
|
pub fn attributes(path: []const u8) ?Attributes {
|
||||||
|
var file = open(path, .{}) orelse return null;
|
||||||
|
defer file.close();
|
||||||
|
return file.attributes();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `path` resolves — handy as a readiness check (e.g. waiting for a mount
|
||||||
|
/// to come up before writing to it).
|
||||||
|
pub fn exists(path: []const u8) bool {
|
||||||
|
return attributes(path) != null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One entry returned by `Directory.next`.
|
||||||
|
pub const Entry = struct {
|
||||||
|
kind: Kind = .regular,
|
||||||
|
size: u64 = 0,
|
||||||
|
name_buffer: [64]u8 = undefined,
|
||||||
|
name_len: usize = 0,
|
||||||
|
|
||||||
|
pub fn name(self: *const Entry) []const u8 {
|
||||||
|
return self.name_buffer[0..self.name_len];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// An open directory being listed, cursor-advanced by `next`.
|
||||||
|
pub const Directory = struct {
|
||||||
|
node: u64,
|
||||||
|
cursor: u64 = 0,
|
||||||
|
|
||||||
|
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||||
|
/// on error.
|
||||||
|
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||||
|
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||||
|
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &buffer) orelse return false;
|
||||||
|
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||||
|
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||||
|
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||||
|
entry.kind = kindFromWire(header.kind);
|
||||||
|
entry.size = header.size;
|
||||||
|
const source = r.payload[protocol.directory_entry_size..];
|
||||||
|
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||||
|
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||||
|
entry.name_len = nlen;
|
||||||
|
self.cursor += 1;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the VFS's open handle for this directory.
|
||||||
|
pub fn close(self: *Directory) void {
|
||||||
|
var f = File{ .node = self.node };
|
||||||
|
f.close();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||||
|
pub fn openDirectory(path: []const u8) ?Directory {
|
||||||
|
const file = open(path, .{ .directory = true }) orelse return null;
|
||||||
|
return .{ .node = file.node };
|
||||||
|
}
|
||||||
|
|
||||||
|
// A path-based request that returns only a status (mkdir, unlink).
|
||||||
|
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||||
|
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
|
||||||
|
const r = transact(request, path, &.{}) orelse return false;
|
||||||
|
return r.reply.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||||
|
/// success. Only works under a mounted filesystem that supports directories.
|
||||||
|
pub fn makeDirectory(path: []const u8) bool {
|
||||||
|
return pathOperation(.mkdir, path);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||||
|
/// (a separate directory-removal would have to check emptiness).
|
||||||
|
pub fn remove(path: []const u8) bool {
|
||||||
|
return pathOperation(.unlink, path);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
|
||||||
|
/// directory, 8.3-name rename only for now). Returns true on success.
|
||||||
|
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||||
|
const total = old_path.len + 1 + new_path.len;
|
||||||
|
if (total > protocol.maximum_payload) return false;
|
||||||
|
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||||
|
@memcpy(payload[0..old_path.len], old_path);
|
||||||
|
payload[old_path.len] = 0;
|
||||||
|
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
|
||||||
|
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||||
|
const r = transact(request, payload[0..total], &.{}) orelse return false;
|
||||||
|
return r.reply.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||||
|
/// the VFS then routes everything under `target` to that backend. This is the one
|
||||||
|
/// call that hands the VFS a capability (the backend endpoint). Returns true on
|
||||||
|
/// success.
|
||||||
|
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||||
|
const h = vfs() orelse return false;
|
||||||
|
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
|
||||||
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||||
|
const tlen = @min(target.len, protocol.maximum_payload);
|
||||||
|
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
|
||||||
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
|
||||||
|
if (result.len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
@@ -12,6 +12,9 @@
|
|||||||
//! (arguments arrive via `init`).
|
//! (arguments arrive via `init`).
|
||||||
|
|
||||||
pub const system = @import("system.zig");
|
pub const system = @import("system.zig");
|
||||||
|
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||||
|
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||||
|
pub const time = @import("time.zig");
|
||||||
pub const heap = @import("heap.zig");
|
pub const heap = @import("heap.zig");
|
||||||
pub const ipc = @import("ipc.zig");
|
pub const ipc = @import("ipc.zig");
|
||||||
pub const start = @import("start.zig");
|
pub const start = @import("start.zig");
|
||||||
@@ -21,7 +24,7 @@ pub const vfs_protocol = @import("vfs-protocol");
|
|||||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||||
|
|
||||||
/// The power protocol: events (button, lid, battery) + shutdown (docs/m21-plan.md).
|
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||||
pub const power_protocol = @import("power-protocol");
|
pub const power_protocol = @import("power-protocol");
|
||||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||||
/// service. See library/runtime/input.zig and system/services/input/.
|
/// service. See library/runtime/input.zig and system/services/input/.
|
||||||
@@ -35,6 +38,33 @@ pub const device = @import("device.zig");
|
|||||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||||
pub const dma = @import("dma.zig");
|
pub const dma = @import("dma.zig");
|
||||||
|
|
||||||
|
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||||
|
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shm.zig
|
||||||
|
/// and docs/display-v2.md.
|
||||||
|
pub const shm = @import("shm.zig");
|
||||||
|
|
||||||
|
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||||
|
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||||
|
pub const usb = @import("usb.zig");
|
||||||
|
|
||||||
|
/// Block-device client: read/write a block device (a USB stick, via
|
||||||
|
/// usb-storage). See library/runtime/block.zig.
|
||||||
|
pub const block = @import("block.zig");
|
||||||
|
|
||||||
|
/// Display-service client: query the mode, and (from D3) create layers, draw, and
|
||||||
|
/// present frames. See library/runtime/display.zig and system/services/display/.
|
||||||
|
pub const display = @import("display.zig");
|
||||||
|
/// The display wire protocol (shared with the display service and its clients).
|
||||||
|
pub const display_protocol = @import("display-protocol");
|
||||||
|
/// The scanout wire protocol: the compositor's present channel to a native scanout driver
|
||||||
|
/// (virtio-gpu). See system/services/display/scanout-protocol.zig and docs/display-v2.md.
|
||||||
|
pub const scanout_protocol = @import("scanout-protocol");
|
||||||
|
|
||||||
|
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||||
|
/// layer danos programs use directly, and where the operations that later become
|
||||||
|
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||||
|
pub const fs = @import("fs.zig");
|
||||||
|
|
||||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||||
pub const panic = start.panic;
|
pub const panic = start.panic;
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,57 @@
|
|||||||
|
//! User-space shared memory: `shm_create` / `shm_map`. A process creates a shareable,
|
||||||
|
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||||
|
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||||
|
//! `shm_map`s it to map the same physical pages. The kernel primitive under the display
|
||||||
|
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||||
|
//! generalization of capability passing from endpoints to memory objects.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||||
|
/// another process as an `ipc_call` send_cap.
|
||||||
|
pub const Region = struct {
|
||||||
|
ptr: [*]u8,
|
||||||
|
handle: ipc.Handle,
|
||||||
|
len: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||||
|
/// Returns the region or null on failure. Two return values — vaddr in rax, handle in rdx —
|
||||||
|
/// so this is a hand-written stub like `dma.alloc`.
|
||||||
|
pub fn create(len: usize) ?Region {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined; // out: the capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shm_create)),
|
||||||
|
[a0] "{rdi}" (len),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map the shared region named by a capability `handle` this process received (via an
|
||||||
|
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||||
|
/// Returns the pointer, or null on failure.
|
||||||
|
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||||
|
const r = sc.systemCall1(.shm_map, handle);
|
||||||
|
if (failed(r)) return null;
|
||||||
|
return @ptrFromInt(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||||
|
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||||
|
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||||
|
/// `attach_backing`. Returns null on failure.
|
||||||
|
pub fn physical(handle: ipc.Handle) ?usize {
|
||||||
|
const r = sc.systemCall1(.shm_physical, handle);
|
||||||
|
if (failed(r)) return null;
|
||||||
|
return r;
|
||||||
|
}
|
||||||
@@ -51,6 +51,24 @@ pub fn clock() u64 {
|
|||||||
return @intCast(sc.systemCall0(.clock));
|
return @intCast(sc.systemCall0(.clock));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
|
||||||
|
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
|
||||||
|
/// filesystem stamps as a file's modification time. Formatting it into a calendar
|
||||||
|
/// date/timezone is user-space policy layered on top.
|
||||||
|
pub fn wallClock() u64 {
|
||||||
|
return @intCast(sc.systemCall0(.wall_clock));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy bytes out of the kernel's in-memory diagnostic log — the accumulated
|
||||||
|
/// stream of everything `write` (and the kernel itself) has emitted — starting at
|
||||||
|
/// `offset`, into `out`. Returns the number of bytes copied (0 at end of buffer).
|
||||||
|
/// A program reads the whole log by looping from offset 0, advancing by the return
|
||||||
|
/// value, until it gets 0. This is how the boot log is persisted to disk on a
|
||||||
|
/// headless/real machine where serial output is otherwise lost.
|
||||||
|
pub fn klogRead(offset: usize, out: []u8) usize {
|
||||||
|
return sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||||
|
}
|
||||||
|
|
||||||
/// End the process. Never returns.
|
/// End the process. Never returns.
|
||||||
pub fn exit(code: usize) noreturn {
|
pub fn exit(code: usize) noreturn {
|
||||||
_ = sc.systemCall1(.exit, code);
|
_ = sc.systemCall1(.exit, code);
|
||||||
|
|||||||
@@ -0,0 +1,169 @@
|
|||||||
|
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||||
|
//!
|
||||||
|
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||||
|
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||||
|
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||||
|
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||||
|
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||||
|
//!
|
||||||
|
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||||
|
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||||
|
//! CLOCK_REALTIME) layered on top later.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
const nanos_per_micro: u64 = 1_000;
|
||||||
|
const nanos_per_milli: u64 = 1_000_000;
|
||||||
|
const nanos_per_second: u64 = 1_000_000_000;
|
||||||
|
|
||||||
|
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||||
|
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||||
|
/// kernel's millisecond granularity and rounding down could return early.
|
||||||
|
pub const Duration = struct {
|
||||||
|
ns: u64,
|
||||||
|
|
||||||
|
pub fn fromNanos(n: u64) Duration {
|
||||||
|
return .{ .ns = n };
|
||||||
|
}
|
||||||
|
pub fn fromMicros(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_micro };
|
||||||
|
}
|
||||||
|
pub fn fromMillis(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_milli };
|
||||||
|
}
|
||||||
|
pub fn fromSeconds(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_second };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn asNanos(d: Duration) u64 {
|
||||||
|
return d.ns;
|
||||||
|
}
|
||||||
|
pub fn asMicros(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_micro;
|
||||||
|
}
|
||||||
|
pub fn asMillis(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_milli;
|
||||||
|
}
|
||||||
|
pub fn asSeconds(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_second;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||||
|
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||||
|
pub fn ceilMillis(d: Duration) u64 {
|
||||||
|
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn plus(a: Duration, b: Duration) Duration {
|
||||||
|
return .{ .ns = a.ns +| b.ns };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||||
|
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||||
|
/// saturate at zero rather than wrap.
|
||||||
|
pub const Instant = struct {
|
||||||
|
ns: u64,
|
||||||
|
|
||||||
|
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||||
|
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||||
|
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||||
|
return .{ .ns = self.ns -| earlier.ns };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How long since this instant, sampled now.
|
||||||
|
pub fn elapsed(self: Instant) Duration {
|
||||||
|
return now().since(self);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||||
|
pub fn plus(self: Instant, d: Duration) Instant {
|
||||||
|
return .{ .ns = self.ns +| d.ns };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||||
|
pub fn reached(deadline: Instant) bool {
|
||||||
|
return now().ns >= deadline.ns;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The current monotonic time.
|
||||||
|
pub fn now() Instant {
|
||||||
|
return .{ .ns = system.clock() };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||||
|
/// want a plain integer instead of an `Instant`.
|
||||||
|
pub fn monotonicNanos() u64 {
|
||||||
|
return system.clock();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||||
|
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||||
|
/// "unavailable" instead of assuming the clock advances.
|
||||||
|
pub fn available() bool {
|
||||||
|
return system.clock() != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||||
|
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||||
|
/// `spin`.
|
||||||
|
pub fn sleep(d: Duration) void {
|
||||||
|
system.sleep(d.ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||||
|
pub fn sleepMillis(ms: u64) void {
|
||||||
|
system.sleep(ms);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||||
|
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||||
|
/// Prefer `sleep` for anything at or above a millisecond.
|
||||||
|
pub fn spin(d: Duration) void {
|
||||||
|
const deadline = now().plus(d);
|
||||||
|
while (!deadline.reached()) {}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||||
|
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||||
|
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||||
|
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||||
|
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||||
|
pub fn after(endpoint: usize, d: Duration) bool {
|
||||||
|
return system.timerOnce(endpoint, d.ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "Duration unit conversions round toward zero" {
|
||||||
|
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||||
|
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||||
|
const t0 = Instant{ .ns = 1_000 };
|
||||||
|
const t1 = Instant{ .ns = 4_000 };
|
||||||
|
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||||
|
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||||
|
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||||
|
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||||
|
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||||
|
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||||
|
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||||
|
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||||
|
}
|
||||||
@@ -0,0 +1,162 @@
|
|||||||
|
//! USB class-driver client: the helper a keyboard, mouse, or mass-storage driver
|
||||||
|
//! uses to reach its device through the xHCI bus driver, so it never hand-rolls
|
||||||
|
//! the transfer-protocol IPC. Layered over `ipc` and the shared
|
||||||
|
//! `usb-transfer-protocol` wire format, the way `input.zig` layers over the input
|
||||||
|
//! service and `device.zig` over the raw device calls.
|
||||||
|
//!
|
||||||
|
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||||
|
//! if (!usb.helloManager(id)) return; // meet the spawn deadline
|
||||||
|
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||||
|
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||||
|
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||||
|
//! while (true) { ... ipc.replyWait(device.endpoint, ...) ... } // its own loop
|
||||||
|
//!
|
||||||
|
//! Reports are delivered to `device.endpoint` as asynchronous `InterruptReport`
|
||||||
|
//! messages (the class driver runs a bare `replyWait` loop to read them, because
|
||||||
|
//! the service harness drops buffered-message payloads — see service.zig).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("usb-transfer-protocol");
|
||||||
|
const device_manager = @import("device-manager-protocol");
|
||||||
|
|
||||||
|
pub const Endpoint = protocol.Endpoint;
|
||||||
|
pub const InterruptReport = protocol.InterruptReport;
|
||||||
|
pub const max_report_data = protocol.max_report_data;
|
||||||
|
|
||||||
|
// Endpoint transfer types (EndpointDescriptor attributes), for `findEndpoint`.
|
||||||
|
pub const transfer_type_bulk: u8 = 2;
|
||||||
|
pub const transfer_type_interrupt: u8 = 3;
|
||||||
|
|
||||||
|
/// An opened USB device: the bus endpoint to send requests to, this driver's own
|
||||||
|
/// endpoint that reports arrive on, the device token, and the interface's
|
||||||
|
/// endpoints (so a driver need not re-read the configuration descriptor).
|
||||||
|
pub const Device = struct {
|
||||||
|
bus: ipc.Handle,
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
token: u64,
|
||||||
|
class: u8,
|
||||||
|
subclass: u8,
|
||||||
|
protocol_code: u8,
|
||||||
|
interface_number: u8,
|
||||||
|
endpoint_count: usize = 0,
|
||||||
|
endpoints: [protocol.max_reported_endpoints]Endpoint = undefined,
|
||||||
|
|
||||||
|
/// The interface's first endpoint of the given transfer type and direction
|
||||||
|
/// (`transfer_type_bulk` / `transfer_type_interrupt`), or null.
|
||||||
|
pub fn findEndpoint(self: *const Device, transfer_type: u8, direction_in: bool) ?Endpoint {
|
||||||
|
for (self.endpoints[0..self.endpoint_count]) |endpoint| {
|
||||||
|
if (endpoint.transfer_type == transfer_type and (endpoint.address & 0x80 != 0) == direction_in) return endpoint;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||||
|
var request = protocol.ControlRequest{
|
||||||
|
.device_token = self.token,
|
||||||
|
.setup = setup,
|
||||||
|
.direction_in = @intFromBool(direction_in),
|
||||||
|
.data_length = @intCast(data.len),
|
||||||
|
};
|
||||||
|
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||||
|
var reply: [@sizeOf(protocol.ControlReply)]u8 = undefined;
|
||||||
|
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||||
|
if (length < @sizeOf(protocol.ControlReply)) return null;
|
||||||
|
const control_reply = std.mem.bytesToValue(protocol.ControlReply, reply[0..@sizeOf(protocol.ControlReply)]);
|
||||||
|
if (control_reply.status != 0) return null;
|
||||||
|
const actual = @min(control_reply.actual_length, data.len);
|
||||||
|
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||||
|
return actual;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A control transfer with no data stage (SET_PROTOCOL, SET_IDLE, ...). The
|
||||||
|
/// `setup` is a bit-cast `usb_abi.Request`.
|
||||||
|
pub fn controlOut(self: *Device, setup: [8]u8) bool {
|
||||||
|
return self.controlTransfer(setup, false, &.{}) != null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A device-to-host control transfer, returning the bytes read into `out`.
|
||||||
|
pub fn controlIn(self: *Device, setup: [8]u8, out: []u8) ?usize {
|
||||||
|
return self.controlTransfer(setup, true, out);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||||
|
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||||
|
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||||
|
var request = protocol.InterruptSubscribeRequest{
|
||||||
|
.device_token = self.token,
|
||||||
|
.endpoint_address = endpoint_address,
|
||||||
|
.max_length = max_length,
|
||||||
|
};
|
||||||
|
var reply: [@sizeOf(protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||||
|
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (length < @sizeOf(protocol.InterruptSubscribeReply)) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.InterruptSubscribeReply, reply[0..@sizeOf(protocol.InterruptSubscribeReply)]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||||
|
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||||
|
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||||
|
var request = protocol.BulkRequest{
|
||||||
|
.device_token = self.token,
|
||||||
|
.physical_address = physical,
|
||||||
|
.length = length,
|
||||||
|
.endpoint_address = endpoint_address,
|
||||||
|
};
|
||||||
|
var reply: [@sizeOf(protocol.BulkReply)]u8 = undefined;
|
||||||
|
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||||
|
if (replied < @sizeOf(protocol.BulkReply)) return null;
|
||||||
|
const bulk_reply = std.mem.bytesToValue(protocol.BulkReply, reply[0..@sizeOf(protocol.BulkReply)]);
|
||||||
|
if (bulk_reply.status != 0) return null;
|
||||||
|
return bulk_reply.actual_length;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Look up the USB bus and open the device with the assigned id, handing over a
|
||||||
|
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
|
||||||
|
/// bus is still coming up (a class driver races the bus driver at boot).
|
||||||
|
pub fn open(device_id: u64) ?Device {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
const bus = while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||||
|
system.sleep(20);
|
||||||
|
} else return null;
|
||||||
|
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||||
|
var request = protocol.OpenRequest{ .device_id = device_id };
|
||||||
|
var reply: [@sizeOf(protocol.OpenReply)]u8 = undefined;
|
||||||
|
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||||
|
if (result.len < @sizeOf(protocol.OpenReply)) return null;
|
||||||
|
const open_reply = std.mem.bytesToValue(protocol.OpenReply, reply[0..@sizeOf(protocol.OpenReply)]);
|
||||||
|
if (open_reply.status != 0) return null;
|
||||||
|
|
||||||
|
var device = Device{
|
||||||
|
.bus = bus,
|
||||||
|
.endpoint = endpoint,
|
||||||
|
.token = open_reply.device_token,
|
||||||
|
.class = open_reply.interface_class,
|
||||||
|
.subclass = open_reply.interface_subclass,
|
||||||
|
.protocol_code = open_reply.interface_protocol,
|
||||||
|
.interface_number = open_reply.interface_number,
|
||||||
|
.endpoint_count = @min(open_reply.endpoint_count, protocol.max_reported_endpoints),
|
||||||
|
};
|
||||||
|
for (0..device.endpoint_count) |index| device.endpoints[index] = open_reply.endpoints[index];
|
||||||
|
return device;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hello the device manager as a class driver (Role.device) so a supervised
|
||||||
|
/// spawn meets its hello deadline. Retries while the manager comes up.
|
||||||
|
pub fn helloManager(device_id: u64) bool {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
const manager = while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||||
|
system.sleep(20);
|
||||||
|
} else return false;
|
||||||
|
|
||||||
|
const hello = device_manager.Hello{ .role = @intFromEnum(device_manager.Role.device), .device_id = device_id };
|
||||||
|
var reply: [device_manager.message_maximum]u8 = undefined;
|
||||||
|
const length = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch return false;
|
||||||
|
if (length < device_manager.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(device_manager.HelloReply, reply[0..device_manager.reply_size]).status == 0;
|
||||||
|
}
|
||||||
+12
-1
@@ -58,6 +58,11 @@ pub const SystemCall = enum(u64) {
|
|||||||
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||||
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
|
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||||
|
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||||
|
shm_create = 34, // shm_create(len) -> vaddr (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||||
|
shm_map = 35, // shm_map(cap) -> vaddr: map the shared region named by a received capability into this AS (the same physical pages the creator sees)
|
||||||
|
shm_physical = 36, // shm_physical(cap) -> paddr: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -177,7 +182,13 @@ pub const ServiceId = enum(u32) {
|
|||||||
input = 2,
|
input = 2,
|
||||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/m21-plan.md; domain-named per decision 7 — the acpi service registers it on x86, a PSCI service will on ARM)
|
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||||
|
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||||
|
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||||
|
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||||
|
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||||
|
shm_test = 10, // the shm test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||||
|
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+32
-66
@@ -3,11 +3,13 @@
|
|||||||
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||||
//! bootloader handed us) and translates the static tables into the generic
|
//! bootloader handed us) and translates the static tables into the generic
|
||||||
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||||
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
//! source. This is deliberately the *static-table* path, and **only** that: MADT
|
||||||
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
|
||||||
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
|
||||||
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
//! interpreted here — the kernel collects the blobs and publishes them on the
|
||||||
//! interpretation is a separate, larger subproject.
|
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
|
||||||
|
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
|
||||||
|
//! off the single-core critical path (nothing else runs alongside it there).
|
||||||
//!
|
//!
|
||||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||||
@@ -19,7 +21,6 @@ const boot_handoff = @import("boot-handoff");
|
|||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const parameters = @import("parameters");
|
const parameters = @import("parameters");
|
||||||
const device_model = @import("device-model.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const aml = @import("aml/aml.zig");
|
|
||||||
const DeviceTree = device_model.DeviceTree;
|
const DeviceTree = device_model.DeviceTree;
|
||||||
const Hal = device_model.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
@@ -37,8 +38,11 @@ pub const RegisterAccess = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
/// The power register map, extracted from the FADT during discovery. Populated by
|
||||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
|
||||||
|
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
|
||||||
|
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
|
||||||
|
/// writes the PM1 control register itself). So this holds only the FADT scalars.
|
||||||
pub const PowerInformation = struct {
|
pub const PowerInformation = struct {
|
||||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||||
@@ -54,10 +58,6 @@ pub const PowerInformation = struct {
|
|||||||
reset: RegisterAccess = .{},
|
reset: RegisterAccess = .{},
|
||||||
reset_value: u8 = 0,
|
reset_value: u8 = 0,
|
||||||
reset_supported: bool = false,
|
reset_supported: bool = false,
|
||||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
|
||||||
/// packages.
|
|
||||||
s5: ?aml.SleepType = null,
|
|
||||||
s3: ?aml.SleepType = null,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||||
@@ -141,32 +141,20 @@ const maximum_cpus = parameters.maximum_cpus;
|
|||||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||||
pub var cpu_information: CpuInformation = .{};
|
pub var cpu_information: CpuInformation = .{};
|
||||||
|
|
||||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
|
||||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
|
||||||
pub const AmlStats = struct {
|
|
||||||
nodes: usize = 0,
|
|
||||||
consumed: usize = 0,
|
|
||||||
total: usize = 0,
|
|
||||||
};
|
|
||||||
pub var aml_stats: AmlStats = .{};
|
|
||||||
|
|
||||||
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
|
||||||
/// device enumeration later. Null until `discover` runs successfully.
|
|
||||||
pub var namespace: ?aml.Namespace = null;
|
|
||||||
|
|
||||||
/// Physical address of the DSDT the FADT points at, or 0.
|
/// Physical address of the DSDT the FADT points at, or 0.
|
||||||
pub var dsdt_physical: u64 = 0;
|
pub var dsdt_physical: u64 = 0;
|
||||||
|
|
||||||
/// The FADT itself (physical + length), published on the acpi-tables node so
|
/// The FADT itself (physical + length), published on the acpi-tables node so
|
||||||
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
||||||
/// the event side (docs/m21-plan.md decision 3). Distinguished from the AML
|
/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML
|
||||||
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
||||||
var fadt_physical: u64 = 0;
|
var fadt_physical: u64 = 0;
|
||||||
var fadt_length: u64 = 0;
|
var fadt_length: u64 = 0;
|
||||||
|
|
||||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
// address + length of each table's post-header bytecode. The kernel does not
|
||||||
// for the sleep-state (`_Sx`) packages.
|
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
|
||||||
|
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
|
||||||
var aml_block_physical: [32]u64 = undefined;
|
var aml_block_physical: [32]u64 = undefined;
|
||||||
var aml_block_len: [32]usize = undefined;
|
var aml_block_len: [32]usize = undefined;
|
||||||
var aml_block_count: usize = 0;
|
var aml_block_count: usize = 0;
|
||||||
@@ -373,7 +361,7 @@ const Hpet = extern struct {
|
|||||||
|
|
||||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
|
||||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
if (rsdp_physical == 0) return error.NoRsdp;
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
boot_memory_regions = memory_regions;
|
boot_memory_regions = memory_regions;
|
||||||
@@ -383,8 +371,6 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
fadt_physical = 0;
|
fadt_physical = 0;
|
||||||
fadt_length = 0;
|
fadt_length = 0;
|
||||||
platform_information = .{};
|
platform_information = .{};
|
||||||
aml_stats = .{};
|
|
||||||
namespace = null;
|
|
||||||
dsdt_physical = 0;
|
dsdt_physical = 0;
|
||||||
aml_block_count = 0;
|
aml_block_count = 0;
|
||||||
|
|
||||||
@@ -401,34 +387,20 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
|||||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
|
||||||
// read the sleep types from it.
|
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
|
||||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
// and the power register map. The AML bytecode (device enumeration and the
|
||||||
for (0..aml_block_count) |i| {
|
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
|
||||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
// job: it claims the acpi-tables node published below, parses the same blobs,
|
||||||
}
|
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
|
||||||
const active = blocks[0..aml_block_count];
|
// AML in the kernel keeps boot latency off the critical, single-core path.
|
||||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
|
||||||
namespace = pr.namespace;
|
|
||||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
|
||||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
|
||||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
|
||||||
// The namespace's Device objects are no longer folded into the kernel
|
|
||||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
|
||||||
// (published below), re-parses the same blobs, and registers + reports
|
|
||||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
|
||||||
// \_S5 sleep type above.
|
|
||||||
} else |_| {
|
|
||||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
|
||||||
// whatever the FADT alone provided.
|
|
||||||
}
|
|
||||||
|
|
||||||
// Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as
|
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
// claimant. Kept even when the kernel-side device building (above) retires
|
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
|
||||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
// now that the kernel keeps only the *static* tables for itself.
|
||||||
publishAcpiTablesNode(device_tree) catch {};
|
publishAcpiTablesNode(device_tree) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -448,7 +420,7 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
|||||||
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||||
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||||
// that names them is parsed, so the grant is the whole space — the honest
|
// that names them is parsed, so the grant is the whole space — the honest
|
||||||
// trust boundary of docs/m19-m20-plan.md decision 5.
|
// trust boundary of docs/discovery.md (the acpi service's one trusted node).
|
||||||
_ = node.addResource(.io_port, 0, 1 << 16);
|
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||||
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||||
// 1 and 12, the RTC, …), and the service registers those devices under this
|
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||||
@@ -461,12 +433,6 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
|||||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
|
||||||
pub fn amlDeviceCount() usize {
|
|
||||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
@@ -502,7 +468,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
|||||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||||
parseDmar(hal, header);
|
parseDmar(hal, header);
|
||||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||||
addAmlBlock(sdt_physical);
|
addAmlBlock(sdt_physical);
|
||||||
}
|
}
|
||||||
// Any other signature is recognised but left opaque for now.
|
// Any other signature is recognised but left opaque for now.
|
||||||
@@ -610,7 +576,7 @@ fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade
|
|||||||
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||||
|
|
||||||
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||||
/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR
|
/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR
|
||||||
/// resources, and `device_register` containment demands the bridge own windows
|
/// resources, and `device_register` containment demands the bridge own windows
|
||||||
/// that cover them. Everything the firmware described is "not hole"; the low
|
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||||
/// aperture runs from the end of the described space below 4 GiB up to the
|
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||||
@@ -740,8 +706,8 @@ const fadt_x_pm_tmr_blk = 208; // GAS
|
|||||||
const flag_reset_register_supported = 1 << 10;
|
const flag_reset_register_supported = 1 << 10;
|
||||||
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||||
|
|
||||||
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
/// FADT -> the power register map (into `power_information`) and the DSDT address,
|
||||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
|
||||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
const len: usize = header.length;
|
const len: usize = header.length;
|
||||||
|
|||||||
@@ -56,7 +56,7 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Count the Device objects in a parsed namespace — what the acpi service
|
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||||
/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts
|
/// (docs/discovery.md) reports, and what the kernel's own parse counts
|
||||||
/// so the two can be checked equal across the ring-3 move.
|
/// so the two can be checked equal across the ring-3 move.
|
||||||
pub fn deviceCount(namespace: *const Namespace) usize {
|
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||||
return countKind(namespace.root, .device);
|
return countKind(namespace.root, .device);
|
||||||
|
|||||||
@@ -29,10 +29,22 @@ pub const DeviceClass = enum(u32) {
|
|||||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||||
acpi_device,
|
acpi_device,
|
||||||
/// The ACPI tables themselves, published as one node for the user-space acpi
|
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||||
/// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs,
|
/// service (docs/discovery.md): memory resources over the AML blobs,
|
||||||
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||||
/// The one node whose claimant is trusted to run firmware bytecode.
|
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||||
acpi_tables,
|
acpi_tables,
|
||||||
|
/// One interface of a USB device, registered by the xHCI bus driver. It owns
|
||||||
|
/// no MMIO — it is reached through its controller — so it carries no
|
||||||
|
/// resources; the (class, subclass, protocol) triple that says what it is
|
||||||
|
/// travels in the bus report's identity, not here.
|
||||||
|
usb_device,
|
||||||
|
/// A scanout framebuffer: a linear region of pixel memory the display service
|
||||||
|
/// claims and maps. Unlike the other classes this one is not firmware-discovered
|
||||||
|
/// — the kernel seeds it from the loader's [[boot-handoff]] framebuffer
|
||||||
|
/// (`devices_broker.seedDisplay`). Its one `memory` resource is the framebuffer,
|
||||||
|
/// flagged write-combining; the geometry to interpret it travels in
|
||||||
|
/// `DeviceDescriptor.display`.
|
||||||
|
display,
|
||||||
unknown,
|
unknown,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -54,10 +66,39 @@ pub const ResourceDescriptor = extern struct {
|
|||||||
kind: u64, // a ResourceKind value
|
kind: u64, // a ResourceKind value
|
||||||
start: u64,
|
start: u64,
|
||||||
len: u64,
|
len: u64,
|
||||||
|
/// A bitmask of `resource_flag_*` hints. Zero for a plain register/RAM window;
|
||||||
|
/// the kernel reads it when it maps the resource. Defaulted so every existing
|
||||||
|
/// literal (which never set flags) keeps compiling and lays out identically.
|
||||||
|
flags: u64 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// `ResourceDescriptor.flags`: map this `memory` resource **write-combining** rather
|
||||||
|
/// than strong-uncacheable — for a framebuffer, where batched bursts to pixel memory
|
||||||
|
/// are the whole point (an uncacheable framebuffer blit is glacial). See
|
||||||
|
/// `mmio_map` (system/kernel/process.zig) and `setupPat` (…/x86_64/paging.zig).
|
||||||
|
pub const resource_flag_write_combining: u64 = 1 << 0;
|
||||||
|
|
||||||
pub const maximum_device_resources = 8;
|
pub const maximum_device_resources = 8;
|
||||||
|
|
||||||
|
/// The byte order of a display's pixels — mirrors the loader's `PixelFormat`
|
||||||
|
/// ([[boot-handoff]]) with the same numeric values, but lives here so user space
|
||||||
|
/// (which must never import the loader↔kernel handoff) can name it. Only the two
|
||||||
|
/// linear 32-bpp layouts a console can paint into exist; see docs/gop.md.
|
||||||
|
pub const DisplayFormat = enum(u32) {
|
||||||
|
rgbx = 0, // byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved
|
||||||
|
bgrx = 1, // byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The geometry of a `display` device's framebuffer, carried in its descriptor so a
|
||||||
|
/// claiming driver knows how to interpret the pixel bytes its `memory` resource maps.
|
||||||
|
/// `pitch` is bytes per row (may exceed `width * 4`; see docs/framebuffer.md).
|
||||||
|
pub const DisplayInfo = extern struct {
|
||||||
|
width: u32 = 0, // visible pixels per row
|
||||||
|
height: u32 = 0, // visible rows
|
||||||
|
pitch: u32 = 0, // bytes from one row's start to the next
|
||||||
|
format: u32 = 0, // a DisplayFormat value
|
||||||
|
};
|
||||||
|
|
||||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||||
pub const no_parent: u64 = ~@as(u64, 0);
|
pub const no_parent: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
@@ -87,4 +128,9 @@ pub const DeviceDescriptor = extern struct {
|
|||||||
resource_count: u64,
|
resource_count: u64,
|
||||||
hid: [8]u8,
|
hid: [8]u8,
|
||||||
resources: [maximum_device_resources]ResourceDescriptor,
|
resources: [maximum_device_resources]ResourceDescriptor,
|
||||||
|
// Framebuffer geometry, meaningful only when `class` is `DeviceClass.display`
|
||||||
|
// (zeroed otherwise). Kept here — a class-specific field on the shared descriptor —
|
||||||
|
// the same way `pci_class` is meaningful only for `pci_device` and `hid` only for
|
||||||
|
// `acpi_device`.
|
||||||
|
display: DisplayInfo = .{},
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -22,13 +22,14 @@ pub const Resource = device_model.Resource;
|
|||||||
pub const ResourceKind = device_model.ResourceKind;
|
pub const ResourceKind = device_model.ResourceKind;
|
||||||
pub const Hal = device_model.Hal;
|
pub const Hal = device_model.Hal;
|
||||||
pub const PowerInformation = acpi.PowerInformation;
|
pub const PowerInformation = acpi.PowerInformation;
|
||||||
pub const AmlStats = acpi.AmlStats;
|
|
||||||
pub const PlatformInformation = acpi.PlatformInformation;
|
pub const PlatformInformation = acpi.PlatformInformation;
|
||||||
pub const RegisterAccess = acpi.RegisterAccess;
|
pub const RegisterAccess = acpi.RegisterAccess;
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
pub const IsoEntry = acpi.IsoEntry;
|
||||||
pub const Cpu = acpi.Cpu;
|
pub const Cpu = acpi.Cpu;
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||||
|
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||||
|
/// owned by the ring-3 acpi service), so they are not here.
|
||||||
pub fn powerInformation() PowerInformation {
|
pub fn powerInformation() PowerInformation {
|
||||||
return acpi.power_information;
|
return acpi.power_information;
|
||||||
}
|
}
|
||||||
@@ -39,18 +40,6 @@ pub fn platformInformation() PlatformInformation {
|
|||||||
return acpi.platform_information;
|
return acpi.platform_information;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
|
||||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
|
||||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
|
||||||
/// count against this.
|
|
||||||
pub fn amlDeviceCount() usize {
|
|
||||||
return acpi.amlDeviceCount();
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn amlStats() AmlStats {
|
|
||||||
return acpi.aml_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The usable logical processors discovered during enumeration — one entry per
|
/// The usable logical processors discovered during enumeration — one entry per
|
||||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||||
@@ -92,12 +81,8 @@ pub fn discover(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls. Soft-off
|
||||||
|
/// (S5) is not a kernel operation — the ring-3 acpi service owns it (docs/power.md).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
power.reboot(hal);
|
power.reboot(hal);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Power the machine off (ACPI S5). Never returns on success.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
power.shutdown(hal);
|
|
||||||
}
|
|
||||||
|
|||||||
+10
-63
@@ -1,34 +1,17 @@
|
|||||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
//! Machine reboot: restart via the FADT reset register, with legacy fallbacks.
|
||||||
//!
|
//!
|
||||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
//! Built on the register map `acpi` extracted from the FADT, driven through the
|
||||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
//! injected `Hal` (port I/O and MMIO). Soft-off (ACPI S5) and suspend (S3) are
|
||||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
//! **not** here: they need the AML sleep-state (`_Sx`) values, which the kernel no
|
||||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
//! longer parses — the ring-3 acpi service owns power management (it re-parses the
|
||||||
//!
|
//! blobs and writes the PM1 control register itself). See docs/power.md. Reboot
|
||||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
//! stays in the kernel because it needs no AML — only the FADT reset register and
|
||||||
//! re-initialisation, a milestone of its own.
|
//! the well-known legacy fallbacks — so it survives as a last-resort restart.
|
||||||
|
|
||||||
const acpi = @import("acpi.zig");
|
const acpi = @import("acpi.zig");
|
||||||
const device_model = @import("device-model.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const Hal = device_model.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
|
||||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
|
||||||
|
|
||||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
|
||||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
|
||||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
|
||||||
pub fn enable(hal: Hal) void {
|
|
||||||
const pi = acpi.power_information;
|
|
||||||
if (!pi.pm1a_cnt.present()) return;
|
|
||||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
|
||||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
|
||||||
|
|
||||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
|
||||||
var spins: usize = 0;
|
|
||||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
@@ -48,42 +31,6 @@ pub fn reboot(hal: Hal) void {
|
|||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
|
||||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
enable(hal);
|
|
||||||
const pi = acpi.power_information;
|
|
||||||
const s5 = pi.s5 orelse return;
|
|
||||||
|
|
||||||
if (pi.pm1a_cnt.present()) {
|
|
||||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
|
||||||
}
|
|
||||||
if (pi.pm1b_cnt.present()) {
|
|
||||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
|
||||||
}
|
|
||||||
delay();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
|
||||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
|
||||||
_ = hal;
|
|
||||||
return error.Unsupported;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
|
||||||
/// [12:10], SLP_EN in bit 13.
|
|
||||||
fn sleepValue(slp_typ: u8) u32 {
|
|
||||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
|
||||||
if (register.mmio) {
|
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
|
||||||
return p.*;
|
|
||||||
}
|
|
||||||
return hal.pioRead(register.width, @intCast(register.address));
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||||
if (register.mmio) {
|
if (register.mmio) {
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
@@ -93,8 +40,8 @@ fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
/// A short busy-wait so a reset takes effect before we fall through to the next
|
||||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
/// method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||||
/// from being optimised away.
|
/// from being optimised away.
|
||||||
fn delay() void {
|
fn delay() void {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
|
|||||||
+139
-51
@@ -8,7 +8,7 @@
|
|||||||
//! buffer at any offset, and bitmap bytes are packed structs so no caller ever needs a magic
|
//! buffer at any offset, and bitmap bytes are packed structs so no caller ever needs a magic
|
||||||
//! mask. Class, subclass, and protocol code tables live in usb-ids.zig.
|
//! mask. Class, subclass, and protocol code tables live in usb-ids.zig.
|
||||||
|
|
||||||
const DeviceState = enum(u8) {
|
pub const DeviceState = enum(u8) {
|
||||||
// Immediately after the USB device is attached to the USB system, it is in this state.
|
// Immediately after the USB device is attached to the USB system, it is in this state.
|
||||||
// The USB specifications do not define the state of a USB device that is detached from
|
// The USB specifications do not define the state of a USB device that is detached from
|
||||||
// a USB system.
|
// a USB system.
|
||||||
@@ -47,7 +47,7 @@ const DeviceState = enum(u8) {
|
|||||||
suspended,
|
suspended,
|
||||||
};
|
};
|
||||||
|
|
||||||
const RequestCode = enum(u8) {
|
pub const RequestCode = enum(u8) {
|
||||||
get_status = 0,
|
get_status = 0,
|
||||||
clear_feature = 1,
|
clear_feature = 1,
|
||||||
set_feature = 3,
|
set_feature = 3,
|
||||||
@@ -59,10 +59,15 @@ const RequestCode = enum(u8) {
|
|||||||
get_interface = 10,
|
get_interface = 10,
|
||||||
set_interface = 11,
|
set_interface = 11,
|
||||||
sync_frame = 12,
|
sync_frame = 12,
|
||||||
|
// Non-exhaustive: class-specific requests (HID, mass storage) reuse this byte
|
||||||
|
// field with codes from their own class's namespace — see the class-request
|
||||||
|
// constructors below. Some class codes numerically coincide with a standard
|
||||||
|
// one; the wire byte is what matters, and the constructors set it explicitly.
|
||||||
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
// Direction of an endpoint, from the host's point of view
|
// Direction of an endpoint, from the host's point of view
|
||||||
const EndpointDirection = enum(u1) {
|
pub const EndpointDirection = enum(u1) {
|
||||||
out = 0,
|
out = 0,
|
||||||
in = 1,
|
in = 1,
|
||||||
};
|
};
|
||||||
@@ -74,7 +79,7 @@ const EndpointDirection = enum(u1) {
|
|||||||
|
|
||||||
// The bus address of a device, assigned by the host with SET_ADDRESS. Addresses are 7 bits
|
// The bus address of a device, assigned by the host with SET_ADDRESS. Addresses are 7 bits
|
||||||
// wide.
|
// wide.
|
||||||
const DeviceAddress = enum(u7) {
|
pub const DeviceAddress = enum(u7) {
|
||||||
// The default address every device answers at after a reset, until SET_ADDRESS
|
// The default address every device answers at after a reset, until SET_ADDRESS
|
||||||
// completes
|
// completes
|
||||||
default = 0,
|
default = 0,
|
||||||
@@ -82,7 +87,7 @@ const DeviceAddress = enum(u7) {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Identifies a configuration; from ConfigurationDescriptor.configuration_value.
|
// Identifies a configuration; from ConfigurationDescriptor.configuration_value.
|
||||||
const ConfigurationValue = enum(u8) {
|
pub const ConfigurationValue = enum(u8) {
|
||||||
// Not configured: returned by GET_CONFIGURATION while the device is in the address
|
// Not configured: returned by GET_CONFIGURATION while the device is in the address
|
||||||
// state, and passed to SET_CONFIGURATION to return a configured device to the address
|
// state, and passed to SET_CONFIGURATION to return a configured device to the address
|
||||||
// state
|
// state
|
||||||
@@ -92,11 +97,11 @@ const ConfigurationValue = enum(u8) {
|
|||||||
|
|
||||||
// Identifies an interface within a configuration; from
|
// Identifies an interface within a configuration; from
|
||||||
// InterfaceDescriptor.interface_number.
|
// InterfaceDescriptor.interface_number.
|
||||||
const InterfaceNumber = enum(u8) { _ };
|
pub const InterfaceNumber = enum(u8) { _ };
|
||||||
|
|
||||||
// Selects between the alternate settings of one interface; from
|
// Selects between the alternate settings of one interface; from
|
||||||
// InterfaceDescriptor.alternate_setting.
|
// InterfaceDescriptor.alternate_setting.
|
||||||
const AlternateSetting = enum(u8) {
|
pub const AlternateSetting = enum(u8) {
|
||||||
// The default setting of an interface
|
// The default setting of an interface
|
||||||
default = 0,
|
default = 0,
|
||||||
_,
|
_,
|
||||||
@@ -104,7 +109,7 @@ const AlternateSetting = enum(u8) {
|
|||||||
|
|
||||||
// The number of an endpoint within a device, 4 bits wide. The direction bit carried
|
// The number of an endpoint within a device, 4 bits wide. The direction bit carried
|
||||||
// alongside it tells the two endpoints sharing a number apart.
|
// alongside it tells the two endpoints sharing a number apart.
|
||||||
const EndpointNumber = enum(u4) {
|
pub const EndpointNumber = enum(u4) {
|
||||||
// Endpoint zero: the default control pipe every device provides
|
// Endpoint zero: the default control pipe every device provides
|
||||||
default_control = 0,
|
default_control = 0,
|
||||||
_,
|
_,
|
||||||
@@ -112,7 +117,7 @@ const EndpointNumber = enum(u4) {
|
|||||||
|
|
||||||
// Index of a STRING descriptor, stored in descriptors that reference a string and passed to
|
// Index of a STRING descriptor, stored in descriptors that reference a string and passed to
|
||||||
// GET_DESCRIPTOR to read it.
|
// GET_DESCRIPTOR to read it.
|
||||||
const StringIndex = enum(u8) {
|
pub const StringIndex = enum(u8) {
|
||||||
// The device has no string descriptor for this field
|
// The device has no string descriptor for this field
|
||||||
none = 0,
|
none = 0,
|
||||||
_,
|
_,
|
||||||
@@ -121,7 +126,7 @@ const StringIndex = enum(u8) {
|
|||||||
// Characteristics of a device request (the bmRequestType field of a set-up packet). Fields are
|
// Characteristics of a device request (the bmRequestType field of a set-up packet). Fields are
|
||||||
// declared least-significant first: recipient occupies bits 4...0, kind bits 6...5, and
|
// declared least-significant first: recipient occupies bits 4...0, kind bits 6...5, and
|
||||||
// direction bit 7.
|
// direction bit 7.
|
||||||
const RequestType = packed struct(u8) {
|
pub const RequestType = packed struct(u8) {
|
||||||
// The recipient of the request (values 4...31 are reserved)
|
// The recipient of the request (values 4...31 are reserved)
|
||||||
recipient: Recipient,
|
recipient: Recipient,
|
||||||
// The type of the request
|
// The type of the request
|
||||||
@@ -129,27 +134,27 @@ const RequestType = packed struct(u8) {
|
|||||||
// Data transfer direction. The value of this bit is ignored when length is zero.
|
// Data transfer direction. The value of this bit is ignored when length is zero.
|
||||||
direction: Direction,
|
direction: Direction,
|
||||||
|
|
||||||
const Recipient = enum(u5) {
|
pub const Recipient = enum(u5) {
|
||||||
device = 0,
|
device = 0,
|
||||||
interface = 1,
|
interface = 1,
|
||||||
endpoint = 2,
|
endpoint = 2,
|
||||||
other = 3,
|
other = 3,
|
||||||
};
|
};
|
||||||
|
|
||||||
const Kind = enum(u2) {
|
pub const Kind = enum(u2) {
|
||||||
standard = 0,
|
standard = 0,
|
||||||
class = 1,
|
class = 1,
|
||||||
vendor = 2,
|
vendor = 2,
|
||||||
reserved = 3,
|
reserved = 3,
|
||||||
};
|
};
|
||||||
|
|
||||||
const Direction = enum(u1) {
|
pub const Direction = enum(u1) {
|
||||||
host_to_device = 0,
|
host_to_device = 0,
|
||||||
device_to_host = 1,
|
device_to_host = 1,
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
const Request = extern struct {
|
pub const Request = extern struct {
|
||||||
// Characteristics of the request
|
// Characteristics of the request
|
||||||
request_type: RequestType,
|
request_type: RequestType,
|
||||||
// Specific request
|
// Specific request
|
||||||
@@ -175,7 +180,7 @@ const Request = extern struct {
|
|||||||
// The format of the index field when request_type specifies an endpoint as the
|
// The format of the index field when request_type specifies an endpoint as the
|
||||||
// recipient. The host should always set the direction bit to zero (but the device
|
// recipient. The host should always set the direction bit to zero (but the device
|
||||||
// should accept either value) when the endpoint is part of a control pipe.
|
// should accept either value) when the endpoint is part of a control pipe.
|
||||||
const EndpointIndex = packed struct(u16) {
|
pub const EndpointIndex = packed struct(u16) {
|
||||||
// Endpoint number
|
// Endpoint number
|
||||||
number: EndpointNumber,
|
number: EndpointNumber,
|
||||||
// Reserved (reset to zero)
|
// Reserved (reset to zero)
|
||||||
@@ -188,7 +193,7 @@ const Request = extern struct {
|
|||||||
|
|
||||||
// The format of the index field when request_type specifies an interface as the
|
// The format of the index field when request_type specifies an interface as the
|
||||||
// recipient.
|
// recipient.
|
||||||
const InterfaceIndex = packed struct(u16) {
|
pub const InterfaceIndex = packed struct(u16) {
|
||||||
// Interface number
|
// Interface number
|
||||||
number: u8,
|
number: u8,
|
||||||
// Reserved (reset to zero)
|
// Reserved (reset to zero)
|
||||||
@@ -199,7 +204,7 @@ const Request = extern struct {
|
|||||||
// descriptor type in the high byte, and the descriptor index in the low byte. The index
|
// descriptor type in the high byte, and the descriptor index in the low byte. The index
|
||||||
// is used to select a specific descriptor (only for CONFIGURATION and STRING
|
// is used to select a specific descriptor (only for CONFIGURATION and STRING
|
||||||
// descriptors) when several descriptors of that type are implemented by a device.
|
// descriptors) when several descriptors of that type are implemented by a device.
|
||||||
const DescriptorValue = packed struct(u16) {
|
pub const DescriptorValue = packed struct(u16) {
|
||||||
// Descriptor index
|
// Descriptor index
|
||||||
index: u8 = 0,
|
index: u8 = 0,
|
||||||
// Descriptor type
|
// Descriptor type
|
||||||
@@ -209,7 +214,7 @@ const Request = extern struct {
|
|||||||
|
|
||||||
// Feature selectors, used as the value field of CLEAR_FEATURE and SET_FEATURE requests. The
|
// Feature selectors, used as the value field of CLEAR_FEATURE and SET_FEATURE requests. The
|
||||||
// comment on each value notes the recipient the selector applies to.
|
// comment on each value notes the recipient the selector applies to.
|
||||||
const FeatureSelector = enum(u16) {
|
pub const FeatureSelector = enum(u16) {
|
||||||
// Halts an endpoint (recipient: endpoint)
|
// Halts an endpoint (recipient: endpoint)
|
||||||
endpoint_halt = 0,
|
endpoint_halt = 0,
|
||||||
// Enables or disables the device's remote wakeup capability (recipient: device)
|
// Enables or disables the device's remote wakeup capability (recipient: device)
|
||||||
@@ -223,7 +228,7 @@ const FeatureSelector = enum(u16) {
|
|||||||
// with the test_mode feature selector. Values 06h...3Fh are reserved for standard test
|
// with the test_mode feature selector. Values 06h...3Fh are reserved for standard test
|
||||||
// selectors and C0h...FFh for vendor-specific test modes; all other unlisted values are
|
// selectors and C0h...FFh for vendor-specific test modes; all other unlisted values are
|
||||||
// reserved.
|
// reserved.
|
||||||
const TestMode = enum(u8) {
|
pub const TestMode = enum(u8) {
|
||||||
test_j = 0x01,
|
test_j = 0x01,
|
||||||
test_k = 0x02,
|
test_k = 0x02,
|
||||||
test_se0_nak = 0x03,
|
test_se0_nak = 0x03,
|
||||||
@@ -234,7 +239,7 @@ const TestMode = enum(u8) {
|
|||||||
|
|
||||||
// The two bytes returned by a GET_STATUS request directed at a device. Fields are declared
|
// The two bytes returned by a GET_STATUS request directed at a device. Fields are declared
|
||||||
// least-significant first.
|
// least-significant first.
|
||||||
const DeviceStatus = packed struct(u16) {
|
pub const DeviceStatus = packed struct(u16) {
|
||||||
// Whether the device is currently self-powered (as opposed to bus-powered). This bit
|
// Whether the device is currently self-powered (as opposed to bus-powered). This bit
|
||||||
// cannot be changed with the SET_FEATURE or CLEAR_FEATURE requests.
|
// cannot be changed with the SET_FEATURE or CLEAR_FEATURE requests.
|
||||||
self_powered: bool,
|
self_powered: bool,
|
||||||
@@ -248,7 +253,7 @@ const DeviceStatus = packed struct(u16) {
|
|||||||
|
|
||||||
// The two bytes returned by a GET_STATUS request directed at an endpoint. (A GET_STATUS
|
// The two bytes returned by a GET_STATUS request directed at an endpoint. (A GET_STATUS
|
||||||
// request directed at an interface returns two bytes that are entirely reserved.)
|
// request directed at an interface returns two bytes that are entirely reserved.)
|
||||||
const EndpointStatus = packed struct(u16) {
|
pub const EndpointStatus = packed struct(u16) {
|
||||||
// Whether the endpoint is currently halted. Set with the SET_FEATURE request using the
|
// Whether the endpoint is currently halted. Set with the SET_FEATURE request using the
|
||||||
// endpoint_halt feature selector, and cleared with CLEAR_FEATURE.
|
// endpoint_halt feature selector, and cleared with CLEAR_FEATURE.
|
||||||
halted: bool,
|
halted: bool,
|
||||||
@@ -258,7 +263,7 @@ const EndpointStatus = packed struct(u16) {
|
|||||||
|
|
||||||
// A target for the standard requests that may be directed at the device, an interface, or
|
// A target for the standard requests that may be directed at the device, an interface, or
|
||||||
// an endpoint.
|
// an endpoint.
|
||||||
const Target = union(enum) {
|
pub const Target = union(enum) {
|
||||||
device,
|
device,
|
||||||
interface: InterfaceNumber,
|
interface: InterfaceNumber,
|
||||||
endpoint: Request.EndpointIndex,
|
endpoint: Request.EndpointIndex,
|
||||||
@@ -287,7 +292,7 @@ const Target = union(enum) {
|
|||||||
// Reads the status of the given target: bit-cast the two bytes the device returns into a
|
// Reads the status of the given target: bit-cast the two bytes the device returns into a
|
||||||
// DeviceStatus or an EndpointStatus. (The two bytes returned for an interface are entirely
|
// DeviceStatus or an EndpointStatus. (The two bytes returned for an interface are entirely
|
||||||
// reserved.)
|
// reserved.)
|
||||||
fn getStatus(target: Target) Request {
|
pub fn getStatus(target: Target) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = target.recipient(),
|
.recipient = target.recipient(),
|
||||||
@@ -303,7 +308,7 @@ fn getStatus(target: Target) Request {
|
|||||||
|
|
||||||
// Clears or disables the given feature. A device cannot be taken out of a test mode with
|
// Clears or disables the given feature. A device cannot be taken out of a test mode with
|
||||||
// this request; test_mode is only cleared by cycling power.
|
// this request; test_mode is only cleared by cycling power.
|
||||||
fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
pub fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = target.recipient(),
|
.recipient = target.recipient(),
|
||||||
@@ -319,7 +324,7 @@ fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
|||||||
|
|
||||||
// Sets or enables the given feature. For the test_mode feature selector, use setTestMode
|
// Sets or enables the given feature. For the test_mode feature selector, use setTestMode
|
||||||
// instead: the test selector rides in the high byte of the index field.
|
// instead: the test selector rides in the high byte of the index field.
|
||||||
fn setFeature(feature: FeatureSelector, target: Target) Request {
|
pub fn setFeature(feature: FeatureSelector, target: Target) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = target.recipient(),
|
.recipient = target.recipient(),
|
||||||
@@ -335,7 +340,7 @@ fn setFeature(feature: FeatureSelector, target: Target) Request {
|
|||||||
|
|
||||||
// Puts a hi-speed device into the given test mode: a SET_FEATURE request with the test_mode
|
// Puts a hi-speed device into the given test mode: a SET_FEATURE request with the test_mode
|
||||||
// feature selector and the test selector in the high byte of the index field.
|
// feature selector and the test selector in the high byte of the index field.
|
||||||
fn setTestMode(mode: TestMode) Request {
|
pub fn setTestMode(mode: TestMode) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .device,
|
.recipient = .device,
|
||||||
@@ -352,7 +357,7 @@ fn setTestMode(mode: TestMode) Request {
|
|||||||
// Assigns the device its bus address, moving it from the default state to the address
|
// Assigns the device its bus address, moving it from the default state to the address
|
||||||
// state. The device does not answer at the new address until the status stage of this
|
// state. The device does not answer at the new address until the status stage of this
|
||||||
// request completes.
|
// request completes.
|
||||||
fn setAddress(address: DeviceAddress) Request {
|
pub fn setAddress(address: DeviceAddress) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .device,
|
.recipient = .device,
|
||||||
@@ -372,7 +377,7 @@ fn setAddress(address: DeviceAddress) Request {
|
|||||||
// - language_id selects the language of a string descriptor, and is zero otherwise.
|
// - language_id selects the language of a string descriptor, and is zero otherwise.
|
||||||
// - length is the number of bytes to read; a device never returns more than length bytes,
|
// - length is the number of bytes to read; a device never returns more than length bytes,
|
||||||
// but may return less if the descriptor is shorter.
|
// but may return less if the descriptor is shorter.
|
||||||
fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
pub fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .device,
|
.recipient = .device,
|
||||||
@@ -389,7 +394,7 @@ fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, l
|
|||||||
// Updates an existing descriptor or adds a new one (optional; many devices do not support
|
// Updates an existing descriptor or adds a new one (optional; many devices do not support
|
||||||
// this request). The parameters mirror getDescriptor; the descriptor itself is sent in the
|
// this request). The parameters mirror getDescriptor; the descriptor itself is sent in the
|
||||||
// DATA stage.
|
// DATA stage.
|
||||||
fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
pub fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .device,
|
.recipient = .device,
|
||||||
@@ -405,7 +410,7 @@ fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, l
|
|||||||
|
|
||||||
// Reads the currently active configuration: @enumFromInt the byte the device returns into a
|
// Reads the currently active configuration: @enumFromInt the byte the device returns into a
|
||||||
// ConfigurationValue, which is none while the device is not configured.
|
// ConfigurationValue, which is none while the device is not configured.
|
||||||
fn getConfiguration() Request {
|
pub fn getConfiguration() Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .device,
|
.recipient = .device,
|
||||||
@@ -422,7 +427,7 @@ fn getConfiguration() Request {
|
|||||||
// Selects the configuration with the given configuration_value (from
|
// Selects the configuration with the given configuration_value (from
|
||||||
// ConfigurationDescriptor.configuration_value), moving the device from the address state to
|
// ConfigurationDescriptor.configuration_value), moving the device from the address state to
|
||||||
// the configured state. Selecting none returns the device to the address state.
|
// the configured state. Selecting none returns the device to the address state.
|
||||||
fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
pub fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .device,
|
.recipient = .device,
|
||||||
@@ -438,7 +443,7 @@ fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
|||||||
|
|
||||||
// Reads the alternate setting currently selected for the given interface: @enumFromInt the
|
// Reads the alternate setting currently selected for the given interface: @enumFromInt the
|
||||||
// byte the device returns into an AlternateSetting.
|
// byte the device returns into an AlternateSetting.
|
||||||
fn getInterface(interface: InterfaceNumber) Request {
|
pub fn getInterface(interface: InterfaceNumber) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .interface,
|
.recipient = .interface,
|
||||||
@@ -454,7 +459,7 @@ fn getInterface(interface: InterfaceNumber) Request {
|
|||||||
|
|
||||||
// Selects an alternate setting (from InterfaceDescriptor.alternate_setting) for the given
|
// Selects an alternate setting (from InterfaceDescriptor.alternate_setting) for the given
|
||||||
// interface.
|
// interface.
|
||||||
fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting) Request {
|
pub fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .interface,
|
.recipient = .interface,
|
||||||
@@ -470,7 +475,7 @@ fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting)
|
|||||||
|
|
||||||
// Reads the two-byte number of the frame in which the given isochronous endpoint's
|
// Reads the two-byte number of the frame in which the given isochronous endpoint's
|
||||||
// repeating pattern of transfers begins.
|
// repeating pattern of transfers begins.
|
||||||
fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
pub fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
||||||
return .{
|
return .{
|
||||||
.request_type = .{
|
.request_type = .{
|
||||||
.recipient = .endpoint,
|
.recipient = .endpoint,
|
||||||
@@ -484,7 +489,79 @@ fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
const DescriptorType = enum(u8) {
|
// Class-specific requests. These carry a `kind = .class` request_type and a
|
||||||
|
// request_code from the interface's class namespace (not the standard
|
||||||
|
// RequestCode set above); the code is written into the same byte field, which
|
||||||
|
// is why RequestCode is non-exhaustive. Each is directed at an interface, whose
|
||||||
|
// number rides in the index field.
|
||||||
|
|
||||||
|
// The HID class request codes (USB HID 1.11 §7.2). Only the ones danos issues
|
||||||
|
// are named; the field on the wire is the raw byte.
|
||||||
|
pub const HidRequestCode = enum(u8) {
|
||||||
|
get_report = 0x01,
|
||||||
|
get_idle = 0x02,
|
||||||
|
get_protocol = 0x03,
|
||||||
|
set_report = 0x09,
|
||||||
|
set_idle = 0x0A,
|
||||||
|
set_protocol = 0x0B,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The two protocols a boot-capable HID device can run (USB HID 1.11 §7.2.5).
|
||||||
|
// A driver selects `boot` for the simplified fixed-format boot report, usable
|
||||||
|
// before a full report-descriptor parser exists.
|
||||||
|
pub const HidProtocol = enum(u8) {
|
||||||
|
boot = 0,
|
||||||
|
report = 1,
|
||||||
|
};
|
||||||
|
|
||||||
|
// SET_PROTOCOL: choose the boot or report protocol on a HID interface.
|
||||||
|
pub fn setProtocol(interface: InterfaceNumber, protocol: HidProtocol) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .host_to_device },
|
||||||
|
.request_code = @enumFromInt(@intFromEnum(HidRequestCode.set_protocol)),
|
||||||
|
.value = @intFromEnum(protocol),
|
||||||
|
.index = @intFromEnum(interface),
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// SET_IDLE: bound a HID interface's report rate. `duration` is in 4 ms units
|
||||||
|
// (0 means report only on change); `report_id` selects a report (0 = all).
|
||||||
|
pub fn setIdle(interface: InterfaceNumber, duration: u8, report_id: u8) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .host_to_device },
|
||||||
|
.request_code = @enumFromInt(@intFromEnum(HidRequestCode.set_idle)),
|
||||||
|
.value = (@as(u16, duration) << 8) | report_id,
|
||||||
|
.index = @intFromEnum(interface),
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Bulk-Only Mass Storage Reset (USB MSC BOT §3.1): ready a mass-storage
|
||||||
|
// interface for the next Command Block Wrapper after a protocol error.
|
||||||
|
pub fn bulkOnlyMassStorageReset(interface: InterfaceNumber) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .host_to_device },
|
||||||
|
.request_code = @enumFromInt(0xFF),
|
||||||
|
.value = 0,
|
||||||
|
.index = @intFromEnum(interface),
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get Max LUN (USB MSC BOT §3.2): read the highest logical unit number the
|
||||||
|
// device supports (0 for a single-LUN flash drive). One byte is returned.
|
||||||
|
pub fn getMaxLun(interface: InterfaceNumber) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .device_to_host },
|
||||||
|
.request_code = @enumFromInt(0xFE),
|
||||||
|
.value = 0,
|
||||||
|
.index = @intFromEnum(interface),
|
||||||
|
.length = 1,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const DescriptorType = enum(u8) {
|
||||||
device = 1,
|
device = 1,
|
||||||
configuration = 2,
|
configuration = 2,
|
||||||
string = 3,
|
string = 3,
|
||||||
@@ -496,7 +573,7 @@ const DescriptorType = enum(u8) {
|
|||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
const DeviceDescriptor = extern struct {
|
pub const DeviceDescriptor = extern struct {
|
||||||
// Size of this descriptor in bytes
|
// Size of this descriptor in bytes
|
||||||
length: u8,
|
length: u8,
|
||||||
// DEVICE Descriptor Type
|
// DEVICE Descriptor Type
|
||||||
@@ -541,7 +618,7 @@ const DeviceDescriptor = extern struct {
|
|||||||
configuration_count: u8,
|
configuration_count: u8,
|
||||||
};
|
};
|
||||||
|
|
||||||
const DeviceQualifierDescriptor = extern struct {
|
pub const DeviceQualifierDescriptor = extern struct {
|
||||||
// Size of this descriptor in bytes
|
// Size of this descriptor in bytes
|
||||||
length: u8,
|
length: u8,
|
||||||
// DEVICE_QUALIFIER Descriptor Type
|
// DEVICE_QUALIFIER Descriptor Type
|
||||||
@@ -564,7 +641,7 @@ const DeviceQualifierDescriptor = extern struct {
|
|||||||
reserved: u8,
|
reserved: u8,
|
||||||
};
|
};
|
||||||
|
|
||||||
const ConfigurationDescriptor = extern struct {
|
pub const ConfigurationDescriptor = extern struct {
|
||||||
// Size of this descriptor in bytes
|
// Size of this descriptor in bytes
|
||||||
length: u8,
|
length: u8,
|
||||||
// CONFIGURATION Descriptor Type
|
// CONFIGURATION Descriptor Type
|
||||||
@@ -593,7 +670,7 @@ const ConfigurationDescriptor = extern struct {
|
|||||||
max_power: u8,
|
max_power: u8,
|
||||||
|
|
||||||
// Configuration characteristics. Fields are declared least-significant first.
|
// Configuration characteristics. Fields are declared least-significant first.
|
||||||
const Attributes = packed struct(u8) {
|
pub const Attributes = packed struct(u8) {
|
||||||
// Reserved, reset to zero (D4...0)
|
// Reserved, reset to zero (D4...0)
|
||||||
reserved: u5,
|
reserved: u5,
|
||||||
// Whether Remote Wakeup is supported by this configuration (D5)
|
// Whether Remote Wakeup is supported by this configuration (D5)
|
||||||
@@ -612,9 +689,9 @@ const ConfigurationDescriptor = extern struct {
|
|||||||
// its alternative speed. The structure of the OTHER_SPEED_CONFIGURATION is identical to that
|
// its alternative speed. The structure of the OTHER_SPEED_CONFIGURATION is identical to that
|
||||||
// of the CONFIGURATION descriptor; the only difference is that the descriptor_type field
|
// of the CONFIGURATION descriptor; the only difference is that the descriptor_type field
|
||||||
// reflects that the descriptor is an OTHER_SPEED_CONFIGURATION descriptor.
|
// reflects that the descriptor is an OTHER_SPEED_CONFIGURATION descriptor.
|
||||||
const OtherSpeedConfigurationDescriptor = ConfigurationDescriptor;
|
pub const OtherSpeedConfigurationDescriptor = ConfigurationDescriptor;
|
||||||
|
|
||||||
const InterfaceDescriptor = extern struct {
|
pub const InterfaceDescriptor = extern struct {
|
||||||
// Size of this descriptor in bytes
|
// Size of this descriptor in bytes
|
||||||
length: u8,
|
length: u8,
|
||||||
// INTERFACE Descriptor Type
|
// INTERFACE Descriptor Type
|
||||||
@@ -654,7 +731,7 @@ const InterfaceDescriptor = extern struct {
|
|||||||
interface_index: StringIndex,
|
interface_index: StringIndex,
|
||||||
};
|
};
|
||||||
|
|
||||||
const EndpointDescriptor = extern struct {
|
pub const EndpointDescriptor = extern struct {
|
||||||
// Size of this descriptor in bytes
|
// Size of this descriptor in bytes
|
||||||
length: u8,
|
length: u8,
|
||||||
// ENDPOINT Descriptor Type
|
// ENDPOINT Descriptor Type
|
||||||
@@ -683,7 +760,7 @@ const EndpointDescriptor = extern struct {
|
|||||||
interval: u8,
|
interval: u8,
|
||||||
|
|
||||||
// The address of an endpoint. Fields are declared least-significant first.
|
// The address of an endpoint. Fields are declared least-significant first.
|
||||||
const Address = packed struct(u8) {
|
pub const Address = packed struct(u8) {
|
||||||
// Endpoint Number (D3...0)
|
// Endpoint Number (D3...0)
|
||||||
number: EndpointNumber,
|
number: EndpointNumber,
|
||||||
// Reserved, reset to zero (D6...4)
|
// Reserved, reset to zero (D6...4)
|
||||||
@@ -693,7 +770,7 @@ const EndpointDescriptor = extern struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// An endpoint's attributes. Fields are declared least-significant first.
|
// An endpoint's attributes. Fields are declared least-significant first.
|
||||||
const Attributes = packed struct(u8) {
|
pub const Attributes = packed struct(u8) {
|
||||||
// Transfer Type (D1...0)
|
// Transfer Type (D1...0)
|
||||||
transfer_type: TransferType,
|
transfer_type: TransferType,
|
||||||
// Synchronization Type; isochronous endpoints only, reserved and reset to zero for
|
// Synchronization Type; isochronous endpoints only, reserved and reset to zero for
|
||||||
@@ -706,21 +783,21 @@ const EndpointDescriptor = extern struct {
|
|||||||
reserved: u2,
|
reserved: u2,
|
||||||
};
|
};
|
||||||
|
|
||||||
const TransferType = enum(u2) {
|
pub const TransferType = enum(u2) {
|
||||||
control = 0,
|
control = 0,
|
||||||
isochronous = 1,
|
isochronous = 1,
|
||||||
bulk = 2,
|
bulk = 2,
|
||||||
interrupt = 3,
|
interrupt = 3,
|
||||||
};
|
};
|
||||||
|
|
||||||
const Synchronization = enum(u2) {
|
pub const Synchronization = enum(u2) {
|
||||||
none = 0,
|
none = 0,
|
||||||
asynchronous = 1,
|
asynchronous = 1,
|
||||||
adaptive = 2,
|
adaptive = 2,
|
||||||
synchronous = 3,
|
synchronous = 3,
|
||||||
};
|
};
|
||||||
|
|
||||||
const Usage = enum(u2) {
|
pub const Usage = enum(u2) {
|
||||||
data = 0,
|
data = 0,
|
||||||
feedback = 1,
|
feedback = 1,
|
||||||
implicit_feedback_data = 2,
|
implicit_feedback_data = 2,
|
||||||
@@ -728,7 +805,7 @@ const EndpointDescriptor = extern struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// The maximum packet size of an endpoint. Fields are declared least-significant first.
|
// The maximum packet size of an endpoint. Fields are declared least-significant first.
|
||||||
const MaxPacketSize = packed struct(u16) {
|
pub const MaxPacketSize = packed struct(u16) {
|
||||||
// Maximum packet size in bytes (bits 10...0)
|
// Maximum packet size in bytes (bits 10...0)
|
||||||
size: u11,
|
size: u11,
|
||||||
// Number of additional transaction opportunities per microframe, for high-speed
|
// Number of additional transaction opportunities per microframe, for high-speed
|
||||||
@@ -739,7 +816,7 @@ const EndpointDescriptor = extern struct {
|
|||||||
reserved: u3,
|
reserved: u3,
|
||||||
};
|
};
|
||||||
|
|
||||||
const AdditionalTransactions = enum(u2) {
|
pub const AdditionalTransactions = enum(u2) {
|
||||||
// None (1 transaction per microframe)
|
// None (1 transaction per microframe)
|
||||||
none = 0,
|
none = 0,
|
||||||
// 1 additional (2 transactions per microframe)
|
// 1 additional (2 transactions per microframe)
|
||||||
@@ -755,7 +832,7 @@ const EndpointDescriptor = extern struct {
|
|||||||
// header, followed by the variable-length payload:
|
// header, followed by the variable-length payload:
|
||||||
// - index 0: an array of two-byte LANGID codes (wLangID[0] through wLangID[x])
|
// - index 0: an array of two-byte LANGID codes (wLangID[0] through wLangID[x])
|
||||||
// - other indices: a Unicode string of N bytes
|
// - other indices: a Unicode string of N bytes
|
||||||
const StringDescriptor = extern struct {
|
pub const StringDescriptor = extern struct {
|
||||||
// Size of this descriptor in bytes
|
// Size of this descriptor in bytes
|
||||||
length: u8,
|
length: u8,
|
||||||
// STRING Descriptor Type
|
// STRING Descriptor Type
|
||||||
@@ -834,7 +911,7 @@ test "bitmap packings match the specification" {
|
|||||||
try expect(hid_type != .device);
|
try expect(hid_type != .device);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn expectRequestBytes(request: Request, expected: [8]u8) !void {
|
pub fn expectRequestBytes(request: Request, expected: [8]u8) !void {
|
||||||
try std.testing.expectEqualSlices(u8, &expected, std.mem.asBytes(&request));
|
try std.testing.expectEqualSlices(u8, &expected, std.mem.asBytes(&request));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -855,3 +932,14 @@ test "standard request constructors encode the specification's set-up packets" {
|
|||||||
try expectRequestBytes(setInterface(@enumFromInt(2), @enumFromInt(1)), .{ 0x01, 11, 1, 0, 2, 0, 0, 0 });
|
try expectRequestBytes(setInterface(@enumFromInt(2), @enumFromInt(1)), .{ 0x01, 11, 1, 0, 2, 0, 0, 0 });
|
||||||
try expectRequestBytes(syncFrame(.{ .number = @enumFromInt(3), .direction = .in }), .{ 0x82, 12, 0, 0, 0x83, 0, 2, 0 });
|
try expectRequestBytes(syncFrame(.{ .number = @enumFromInt(3), .direction = .in }), .{ 0x82, 12, 0, 0, 0x83, 0, 2, 0 });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "class request constructors encode the specification's set-up packets" {
|
||||||
|
// bmRequestType for a host-to-device class request to an interface = 0x21;
|
||||||
|
// device-to-host = 0xA1. The request_code byte is the class code, not a
|
||||||
|
// standard one — SET_PROTOCOL 0x0B, SET_IDLE 0x0A, BOT reset 0xFF, Max LUN 0xFE.
|
||||||
|
try expectRequestBytes(setProtocol(@enumFromInt(0), .boot), .{ 0x21, 0x0B, 0, 0, 0, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(setProtocol(@enumFromInt(1), .report), .{ 0x21, 0x0B, 1, 0, 1, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(setIdle(@enumFromInt(1), 0, 0), .{ 0x21, 0x0A, 0, 0, 1, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(bulkOnlyMassStorageReset(@enumFromInt(0)), .{ 0x21, 0xFF, 0, 0, 0, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(getMaxLun(@enumFromInt(0)), .{ 0xA1, 0xFE, 0, 0, 0, 0, 1, 0 });
|
||||||
|
}
|
||||||
|
|||||||
+62
-19
@@ -11,7 +11,7 @@
|
|||||||
|
|
||||||
// Base class codes (assigned by the USB-IF). The comment on each value notes where the code
|
// Base class codes (assigned by the USB-IF). The comment on each value notes where the code
|
||||||
// may legally appear: in the device descriptor, in interface descriptors, or both.
|
// may legally appear: in the device descriptor, in interface descriptors, or both.
|
||||||
const Class = enum(u8) {
|
pub const Class = enum(u8) {
|
||||||
// Use class information in the interface descriptors (device descriptor only). Each
|
// Use class information in the interface descriptors (device descriptor only). Each
|
||||||
// interface within a configuration specifies its own class information and the various
|
// interface within a configuration specifies its own class information and the various
|
||||||
// interfaces operate independently.
|
// interfaces operate independently.
|
||||||
@@ -72,8 +72,8 @@ const Class = enum(u8) {
|
|||||||
|
|
||||||
// Subclass and protocol codes qualified by Class.hub. Hubs have no subclass codes; the
|
// Subclass and protocol codes qualified by Class.hub. Hubs have no subclass codes; the
|
||||||
// protocol distinguishes the hub's transaction-translator arrangement.
|
// protocol distinguishes the hub's transaction-translator arrangement.
|
||||||
const hub = struct {
|
pub const hub = struct {
|
||||||
const Protocol = enum(u8) {
|
pub const Protocol = enum(u8) {
|
||||||
// Full-speed hub
|
// Full-speed hub
|
||||||
full_speed = 0x00,
|
full_speed = 0x00,
|
||||||
// Hi-speed hub with a single transaction translator
|
// Hi-speed hub with a single transaction translator
|
||||||
@@ -87,8 +87,8 @@ const hub = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Subclass and protocol codes qualified by Class.hid.
|
// Subclass and protocol codes qualified by Class.hid.
|
||||||
const hid = struct {
|
pub const hid = struct {
|
||||||
const SubClass = enum(u8) {
|
pub const SubClass = enum(u8) {
|
||||||
// No subclass
|
// No subclass
|
||||||
none = 0x00,
|
none = 0x00,
|
||||||
// Boot interface: the device also supports the simplified boot protocol, usable by
|
// Boot interface: the device also supports the simplified boot protocol, usable by
|
||||||
@@ -98,7 +98,7 @@ const hid = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Only meaningful when the subclass is boot
|
// Only meaningful when the subclass is boot
|
||||||
const Protocol = enum(u8) {
|
pub const Protocol = enum(u8) {
|
||||||
none = 0x00,
|
none = 0x00,
|
||||||
keyboard = 0x01,
|
keyboard = 0x01,
|
||||||
mouse = 0x02,
|
mouse = 0x02,
|
||||||
@@ -109,8 +109,8 @@ const hid = struct {
|
|||||||
// Subclass and protocol codes qualified by Class.mass_storage. The subclass identifies the
|
// Subclass and protocol codes qualified by Class.mass_storage. The subclass identifies the
|
||||||
// command set the device understands; the protocol identifies the transport used to carry
|
// command set the device understands; the protocol identifies the transport used to carry
|
||||||
// commands, data, and status over the bus.
|
// commands, data, and status over the bus.
|
||||||
const mass_storage = struct {
|
pub const mass_storage = struct {
|
||||||
const SubClass = enum(u8) {
|
pub const SubClass = enum(u8) {
|
||||||
// SCSI command set not reported; de facto, treat as scsi
|
// SCSI command set not reported; de facto, treat as scsi
|
||||||
not_reported = 0x00,
|
not_reported = 0x00,
|
||||||
// Reduced Block Commands: typically flash devices
|
// Reduced Block Commands: typically flash devices
|
||||||
@@ -134,7 +134,7 @@ const mass_storage = struct {
|
|||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
const Protocol = enum(u8) {
|
pub const Protocol = enum(u8) {
|
||||||
// Control/Bulk/Interrupt with command completion interrupt
|
// Control/Bulk/Interrupt with command completion interrupt
|
||||||
cbi_completion_interrupt = 0x00,
|
cbi_completion_interrupt = 0x00,
|
||||||
// Control/Bulk/Interrupt without command completion interrupt
|
// Control/Bulk/Interrupt without command completion interrupt
|
||||||
@@ -152,8 +152,8 @@ const mass_storage = struct {
|
|||||||
// Subclass and protocol codes qualified by Class.communications (CDC). The protocol codes
|
// Subclass and protocol codes qualified by Class.communications (CDC). The protocol codes
|
||||||
// are model-specific; the useful invariant is the subclass, which selects the control model
|
// are model-specific; the useful invariant is the subclass, which selects the control model
|
||||||
// the interface implements.
|
// the interface implements.
|
||||||
const communications = struct {
|
pub const communications = struct {
|
||||||
const SubClass = enum(u8) {
|
pub const SubClass = enum(u8) {
|
||||||
// Direct line control model
|
// Direct line control model
|
||||||
direct_line = 0x01,
|
direct_line = 0x01,
|
||||||
// Abstract control model: USB modems and serial adapters
|
// Abstract control model: USB modems and serial adapters
|
||||||
@@ -185,15 +185,15 @@ const communications = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Subclass and protocol codes qualified by Class.wireless_controller.
|
// Subclass and protocol codes qualified by Class.wireless_controller.
|
||||||
const wireless_controller = struct {
|
pub const wireless_controller = struct {
|
||||||
const SubClass = enum(u8) {
|
pub const SubClass = enum(u8) {
|
||||||
// Radio frequency controllers
|
// Radio frequency controllers
|
||||||
radio_frequency = 0x01,
|
radio_frequency = 0x01,
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
// Only meaningful when the subclass is radio_frequency
|
// Only meaningful when the subclass is radio_frequency
|
||||||
const Protocol = enum(u8) {
|
pub const Protocol = enum(u8) {
|
||||||
// Bluetooth programming interface
|
// Bluetooth programming interface
|
||||||
bluetooth = 0x01,
|
bluetooth = 0x01,
|
||||||
// Ultra-wideband radio control
|
// Ultra-wideband radio control
|
||||||
@@ -207,15 +207,15 @@ const wireless_controller = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Subclass and protocol codes qualified by Class.miscellaneous.
|
// Subclass and protocol codes qualified by Class.miscellaneous.
|
||||||
const miscellaneous = struct {
|
pub const miscellaneous = struct {
|
||||||
const SubClass = enum(u8) {
|
pub const SubClass = enum(u8) {
|
||||||
// Common class
|
// Common class
|
||||||
common = 0x02,
|
common = 0x02,
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
// Only meaningful when the subclass is common
|
// Only meaningful when the subclass is common
|
||||||
const Protocol = enum(u8) {
|
pub const Protocol = enum(u8) {
|
||||||
// Interface association descriptor: at the device level, announces that the
|
// Interface association descriptor: at the device level, announces that the
|
||||||
// configuration groups interfaces into functions with IADs
|
// configuration groups interfaces into functions with IADs
|
||||||
interface_association = 0x01,
|
interface_association = 0x01,
|
||||||
@@ -224,8 +224,8 @@ const miscellaneous = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Subclass and protocol codes qualified by Class.application_specific.
|
// Subclass and protocol codes qualified by Class.application_specific.
|
||||||
const application_specific = struct {
|
pub const application_specific = struct {
|
||||||
const SubClass = enum(u8) {
|
pub const SubClass = enum(u8) {
|
||||||
// Device firmware upgrade
|
// Device firmware upgrade
|
||||||
firmware_upgrade = 0x01,
|
firmware_upgrade = 0x01,
|
||||||
// IrDA bridge
|
// IrDA bridge
|
||||||
@@ -236,6 +236,23 @@ const application_specific = struct {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// Pack a (class, subclass, protocol) triple into one 0xCCSSPP value — the
|
||||||
|
/// bus-native identity a USB bus driver reports in `ChildAdded.identity` and the
|
||||||
|
/// device manager matches on (the USB analog of a packed PCI class code). Mirrors
|
||||||
|
/// `pci_class.ClassCode.pack`, so both sides build/decode the identical u64.
|
||||||
|
pub fn packTriple(class: u8, subclass: u8, protocol: u8) u64 {
|
||||||
|
return (@as(u64, class) << 16) | (@as(u64, subclass) << 8) | protocol;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The inverse of `packTriple`.
|
||||||
|
pub fn unpackTriple(triple: u64) struct { class: u8, subclass: u8, protocol: u8 } {
|
||||||
|
return .{
|
||||||
|
.class = @truncate(triple >> 16),
|
||||||
|
.subclass = @truncate(triple >> 8),
|
||||||
|
.protocol = @truncate(triple),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
test "class codes match the USB-IF assignments" {
|
test "class codes match the USB-IF assignments" {
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const expectEqual = std.testing.expectEqual;
|
const expectEqual = std.testing.expectEqual;
|
||||||
@@ -262,3 +279,29 @@ test "class codes match the USB-IF assignments" {
|
|||||||
_ = miscellaneous.Protocol.interface_association;
|
_ = miscellaneous.Protocol.interface_association;
|
||||||
_ = application_specific.SubClass.firmware_upgrade;
|
_ = application_specific.SubClass.firmware_upgrade;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "packTriple / unpackTriple round-trip the identity a bus driver reports" {
|
||||||
|
const std = @import("std");
|
||||||
|
const expectEqual = std.testing.expectEqual;
|
||||||
|
|
||||||
|
// A boot keyboard interface: HID / boot / keyboard.
|
||||||
|
const keyboard = packTriple(
|
||||||
|
@intFromEnum(Class.hid),
|
||||||
|
@intFromEnum(hid.SubClass.boot),
|
||||||
|
@intFromEnum(hid.Protocol.keyboard),
|
||||||
|
);
|
||||||
|
try expectEqual(@as(u64, 0x03_01_01), keyboard);
|
||||||
|
|
||||||
|
// A flash drive interface: mass storage / SCSI / bulk-only.
|
||||||
|
const storage = packTriple(
|
||||||
|
@intFromEnum(Class.mass_storage),
|
||||||
|
@intFromEnum(mass_storage.SubClass.scsi),
|
||||||
|
@intFromEnum(mass_storage.Protocol.bulk_only),
|
||||||
|
);
|
||||||
|
try expectEqual(@as(u64, 0x08_06_50), storage);
|
||||||
|
|
||||||
|
const parts = unpackTriple(storage);
|
||||||
|
try expectEqual(@as(u8, 0x08), parts.class);
|
||||||
|
try expectEqual(@as(u8, 0x06), parts.subclass);
|
||||||
|
try expectEqual(@as(u8, 0x50), parts.protocol);
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,211 +0,0 @@
|
|||||||
//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one.
|
|
||||||
//!
|
|
||||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
|
||||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
|
||||||
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
|
||||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
|
||||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
|
||||||
//!
|
|
||||||
//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children
|
|
||||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
|
||||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
|
||||||
//! every one of those resources is contained in what `bus` was granted. A comparator
|
|
||||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
|
||||||
//!
|
|
||||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
|
||||||
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
|
||||||
//! arbitrary physical memory.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const runtime = @import("runtime");
|
|
||||||
const device = runtime.device;
|
|
||||||
|
|
||||||
const register_general_cap = 0x000;
|
|
||||||
|
|
||||||
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
|
||||||
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
|
||||||
return .{
|
|
||||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
|
||||||
.start = hpet_base + 0x100 + 0x20 * n,
|
|
||||||
.len = 0x20,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
const n = @min(total, buffer.len);
|
|
||||||
for (buffer[0..n]) |d| {
|
|
||||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
|
||||||
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
|
||||||
for (0..d.resource_count) |j| {
|
|
||||||
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The parent's MMIO resource, and its IRQ if it has one.
|
|
||||||
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
|
||||||
var mmio: device.ResourceDescriptor = undefined;
|
|
||||||
var irq: ?device.ResourceDescriptor = null;
|
|
||||||
for (0..d.resource_count) |j| {
|
|
||||||
const r = d.resources[j];
|
|
||||||
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
|
||||||
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
|
||||||
}
|
|
||||||
return .{ .mmio = mmio, .irq = irq };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
|
||||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
|
||||||
if (d.parent == parent_id) return d.id;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn main() void {
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
|
||||||
_ = runtime.system.write("bus: out of memory\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
const parent = findHpet(buffer) orelse {
|
|
||||||
_ = runtime.system.write("bus: no HPET\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const resource = resourcesOf(parent);
|
|
||||||
|
|
||||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
|
||||||
//
|
|
||||||
// Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk
|
|
||||||
// binary — so hpet may own the HPET already. That's not an error, it's the
|
|
||||||
// capability model working: exit quietly and leave the device to its owner. The
|
|
||||||
// `bus` test spawns bus alone, so there it wins the claim.
|
|
||||||
if (!device.claim(parent.id)) {
|
|
||||||
_ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Enumerate the bus: ask the hardware how many children it has.
|
|
||||||
const base = device.mmioMap(parent.id, 0) orelse {
|
|
||||||
_ = runtime.system.write("bus: mmio_map failed\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
|
||||||
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
|
||||||
|
|
||||||
// Publish one child per comparator, each owning only its own window.
|
|
||||||
var published: u64 = 0;
|
|
||||||
var n: u64 = 0;
|
|
||||||
while (n < n_children) : (n += 1) {
|
|
||||||
var child = std.mem.zeroes(device.DeviceDescriptor);
|
|
||||||
child.class = @intFromEnum(device.DeviceClass.timer);
|
|
||||||
child.pci_class = device.no_pci_class;
|
|
||||||
child.hid_len = 6;
|
|
||||||
child.hid[0..6].* = "hpet-t".*;
|
|
||||||
child.resource_count = 1;
|
|
||||||
child.resources[0] = timerWindow(resource.mmio.start, n);
|
|
||||||
// Comparators share the block's interrupt line; only one child can bind it,
|
|
||||||
// but all of them may legitimately name it.
|
|
||||||
if (resource.irq) |i| {
|
|
||||||
child.resources[child.resource_count] = i;
|
|
||||||
child.resource_count += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (device.register(parent.id, &child) == null) {
|
|
||||||
_ = runtime.system.write("bus: register failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
published += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
// The negative case. A window one byte past the end of the parent's must be
|
|
||||||
// refused — otherwise device_register would be "map any physical page you like".
|
|
||||||
// Confirm the table did not grow, not merely that the call returned null: null
|
|
||||||
// also means NoSpace/BadParent, so a size check is what actually proves the
|
|
||||||
// *containment* rule fired.
|
|
||||||
const before = device.enumerate(buffer);
|
|
||||||
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
|
||||||
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
|
||||||
rogue.resource_count = 1;
|
|
||||||
rogue.resources[0] = .{
|
|
||||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
|
||||||
.start = resource.mmio.start + resource.mmio.len,
|
|
||||||
.len = 0x1000,
|
|
||||||
};
|
|
||||||
if (device.register(parent.id, &rogue) != null) {
|
|
||||||
_ = runtime.system.write("bus: FAIL out-of-window child was accepted\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (device.enumerate(buffer) != before) {
|
|
||||||
_ = runtime.system.write("bus: FAIL rogue child leaked into the table\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// And confirm the children came back with the right parent and a *narrower*
|
|
||||||
// window than the bus — read from the table, not from our own memory.
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
var seen: u64 = 0;
|
|
||||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
|
||||||
if (d.parent != parent.id) continue;
|
|
||||||
const w = d.resources[0];
|
|
||||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
|
||||||
_ = runtime.system.write("bus: FAIL child window is not inside the bus\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
seen += 1;
|
|
||||||
}
|
|
||||||
if (seen != published) {
|
|
||||||
_ = runtime.system.write("bus: FAIL child count mismatch\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
|
||||||
// a different process; here bus plays both parts, which exercises the same path.
|
|
||||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
|
||||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
|
||||||
//
|
|
||||||
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
|
||||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
|
||||||
// The *resource* is narrow even though the page isn't.)
|
|
||||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
|
||||||
_ = runtime.system.write("bus: FAIL no child to claim\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
if (!device.claim(child_id)) {
|
|
||||||
_ = runtime.system.write("bus: FAIL could not claim own child\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
|
||||||
_ = runtime.system.write("bus: FAIL child mmio_map refused\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
|
||||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
|
||||||
if (via_child.* != via_bus.*) {
|
|
||||||
_ = runtime.system.write("bus: FAIL child window does not alias the bus register\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
|
||||||
// kernel. Grab a page, free it, and register through the stale address: if the
|
|
||||||
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
|
||||||
// would triple-fault QEMU and the test would time out instead of printing ok.
|
|
||||||
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
|
||||||
if (!runtime.system.mmapFailed(scratch)) {
|
|
||||||
_ = runtime.system.munmap(scratch, 0x1000);
|
|
||||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
|
||||||
if (device.register(parent.id, descriptor) != null) {
|
|
||||||
_ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = runtime.system.write("bus: ok\n");
|
|
||||||
while (true) runtime.system.sleep(1000);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
|
||||||
comptime {
|
|
||||||
_ = &runtime.start._start;
|
|
||||||
}
|
|
||||||
@@ -1,195 +0,0 @@
|
|||||||
//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to
|
|
||||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
|
||||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
|
||||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
|
||||||
//!
|
|
||||||
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
|
||||||
//! scheduler queue; the core runs other work or idles. That is the point of the
|
|
||||||
//! exercise — a driver is a process that sleeps until its device has something to
|
|
||||||
//! say (see docs/drivers.md).
|
|
||||||
//!
|
|
||||||
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
|
||||||
//! simpler, but level is the discipline every real device line needs, and it forces
|
|
||||||
//! the full cycle to be correct:
|
|
||||||
//!
|
|
||||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
|
||||||
//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
|
||||||
//! hpet irq_ack -> kernel unmasks the GSI
|
|
||||||
//!
|
|
||||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
|
||||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
|
||||||
//!
|
|
||||||
//! Register map (HPET spec 1.0a):
|
|
||||||
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
|
||||||
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
|
||||||
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
|
||||||
//! 0x0F0 MAIN_COUNTER
|
|
||||||
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
|
||||||
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
|
||||||
//! 0x108 TIMER0_COMPARATOR
|
|
||||||
|
|
||||||
const runtime = @import("runtime");
|
|
||||||
const mmio = @import("mmio");
|
|
||||||
const device = runtime.device;
|
|
||||||
const ipc = runtime.ipc;
|
|
||||||
|
|
||||||
const register_general_cap = 0x000;
|
|
||||||
const register_general_configuration = 0x010;
|
|
||||||
const register_int_status = 0x020;
|
|
||||||
const register_main_counter = 0x0F0;
|
|
||||||
const register_timer0_configuration = 0x100;
|
|
||||||
const register_timer0_comparator = 0x108;
|
|
||||||
|
|
||||||
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
|
||||||
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
|
||||||
const tn_int_type_level: u64 = 1 << 1;
|
|
||||||
const tn_int_enb: u64 = 1 << 2;
|
|
||||||
const tn_type_periodic: u64 = 1 << 3;
|
|
||||||
const tn_route_shift = 9;
|
|
||||||
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
|
||||||
|
|
||||||
/// Interrupts to observe before declaring victory.
|
|
||||||
const target_ticks = 5;
|
|
||||||
|
|
||||||
/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio).
|
|
||||||
/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so
|
|
||||||
/// UC writes are already ordered) — no barriers are needed here; the point is the
|
|
||||||
/// typed, arch-portable access every driver should use.
|
|
||||||
inline fn rd(base: usize, off: usize) u64 {
|
|
||||||
return mmio.read(u64, base + off);
|
|
||||||
}
|
|
||||||
inline fn wr(base: usize, off: usize, value: u64) void {
|
|
||||||
mmio.write(u64, base + off, value);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
|
||||||
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
|
||||||
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
|
||||||
|
|
||||||
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
|
||||||
const total = device.enumerate(buffer);
|
|
||||||
const n = @min(total, buffer.len);
|
|
||||||
for (buffer[0..n]) |d| {
|
|
||||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
|
||||||
// Skip comparator children a bus driver may have published below the block
|
|
||||||
// (see system/drivers/bus/bus.zig) — we want the register block itself.
|
|
||||||
if (d.parent != device.no_parent) continue;
|
|
||||||
var mmio_index: ?u64 = null;
|
|
||||||
var irq: ?u64 = null;
|
|
||||||
for (0..d.resource_count) |j| {
|
|
||||||
switch (d.resources[j].kind) {
|
|
||||||
@intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j,
|
|
||||||
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (mmio_index) |m| if (irq) |i| {
|
|
||||||
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
|
||||||
};
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn main() void {
|
|
||||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
|
||||||
_ = runtime.system.write("hpet: out of memory\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
const hpet = findHpet(buffer) orelse {
|
|
||||||
_ = runtime.system.write("hpet: no HPET with an IRQ\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
if (!device.claim(hpet.device_id)) {
|
|
||||||
_ = runtime.system.write("hpet: claim failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
|
||||||
_ = runtime.system.write("hpet: mmio_map failed\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
|
||||||
// to raise exactly this line — the kernel will only bind the one it recorded.
|
|
||||||
const gsi = hpet.gsi;
|
|
||||||
|
|
||||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
|
||||||
_ = runtime.system.write("hpet: create_ipc_endpoint failed\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
|
|
||||||
// --- program the hardware ------------------------------------------------
|
|
||||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
|
||||||
const femtos_per_tick = rd(base, register_general_cap) >> 32;
|
|
||||||
if (femtos_per_tick == 0) {
|
|
||||||
_ = runtime.system.write("hpet: bad HPET period\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
|
||||||
|
|
||||||
// Stop the counter and take the legacy route off while we reconfigure.
|
|
||||||
wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt));
|
|
||||||
|
|
||||||
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
|
||||||
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
|
||||||
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
|
||||||
// timer driver does anyway.
|
|
||||||
var t0 = rd(base, register_timer0_configuration);
|
|
||||||
t0 &= ~(tn_route_mask | tn_type_periodic);
|
|
||||||
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
|
||||||
wr(base, register_timer0_configuration, t0);
|
|
||||||
|
|
||||||
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
|
||||||
wr(base, register_int_status, 1);
|
|
||||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
|
||||||
wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable);
|
|
||||||
|
|
||||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
|
||||||
_ = runtime.system.write("hpet: irq_bind failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
_ = runtime.system.write("hpet: bound, sleeping until the hardware speaks\n");
|
|
||||||
|
|
||||||
// --- the driver loop -----------------------------------------------------
|
|
||||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
|
||||||
// runs only because an interrupt fired.
|
|
||||||
var receive: [64]u8 = undefined;
|
|
||||||
var count: usize = 0;
|
|
||||||
while (count < target_ticks) {
|
|
||||||
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
|
||||||
// next line runs only because the HPET raised its line.
|
|
||||||
const r = ipc.replyWait(endpoint, &.{}, &receive, null);
|
|
||||||
if (!r.isNotification()) continue; // a client request, not our IRQ
|
|
||||||
|
|
||||||
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
|
||||||
// line is still asserted and unmasking would refire immediately.
|
|
||||||
wr(base, register_int_status, 1);
|
|
||||||
count += 1;
|
|
||||||
|
|
||||||
if (count < target_ticks) {
|
|
||||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
|
||||||
} else {
|
|
||||||
// Last one: stop the source rather than re-arming, so the line is left
|
|
||||||
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
|
||||||
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
|
||||||
// the line again a moment later.
|
|
||||||
wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb);
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = runtime.system.write("hpet: irq\n");
|
|
||||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
|
||||||
_ = runtime.system.write("hpet: irq_ack failed\n");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = runtime.system.write("hpet: ok\n");
|
|
||||||
while (true) runtime.system.sleep(1000);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub const panic = runtime.panic;
|
|
||||||
comptime {
|
|
||||||
_ = &runtime.start._start;
|
|
||||||
}
|
|
||||||
@@ -1,12 +1,12 @@
|
|||||||
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
||||||
//! (docs/m19-m20-plan.md, M19). The device manager matches the `pci_host_bridge`
|
//! (docs/discovery.md). The device manager matches the `pci_host_bridge`
|
||||||
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
||||||
//! the same per-device contract as usb-xhci-bus.
|
//! the same per-device contract as usb-xhci-bus.
|
||||||
//!
|
//!
|
||||||
//! M19.1 (this increment): claim the bridge, map its ECAM window (resource 0;
|
//! M19.1 (this increment): claim the bridge, map its ECAM window (resource 0;
|
||||||
//! the bus range and the MMIO apertures follow it), walk every
|
//! the bus range and the MMIO apertures follow it), walk every
|
||||||
//! bus/device/function config header, and log what the walk finds — ending
|
//! bus/device/function config header, and log what the walk finds — ending
|
||||||
//! with "pci-bus: N functions found", which the `pci-scan` scenario compares
|
//! with "/system/drivers/pci-bus: N functions found", which the `pci-scan` scenario compares
|
||||||
//! against the kernel's own enumeration. Registration and reports (M19.2), and
|
//! against the kernel's own enumeration. Registration and reports (M19.2), and
|
||||||
//! the kernel walk's retirement (M19.3), build on this proven-equivalent scan.
|
//! the kernel walk's retirement (M19.3), build on this proven-equivalent scan.
|
||||||
|
|
||||||
@@ -31,9 +31,9 @@ fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
|||||||
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||||
var line: [200]u8 = undefined;
|
var line: [200]u8 = undefined;
|
||||||
const text = if (pif.len != 0)
|
const text = if (pif.len != 0)
|
||||||
std.fmt.bufPrint(&line, "pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||||
else
|
else
|
||||||
std.fmt.bufPrint(&line, "pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||||
_ = runtime.system.write(text);
|
_ = runtime.system.write(text);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -74,37 +74,37 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi
|
|||||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
_ = endpoint;
|
_ = endpoint;
|
||||||
if (!device.claim(bridge_id)) {
|
if (!device.claim(bridge_id)) {
|
||||||
writeLine("pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
writeLine("/system/drivers/pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("pci-bus: out of memory\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: out of memory\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const total = device.enumerate(buffer);
|
const total = device.enumerate(buffer);
|
||||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
if (d.id == bridge_id) break d;
|
if (d.id == bridge_id) break d;
|
||||||
} else {
|
} else {
|
||||||
writeLine("pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
writeLine("/system/drivers/pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
||||||
// range rides beside it. The MMIO apertures (M19.0) come after both.
|
// range rides beside it. The MMIO apertures (M19.0) come after both.
|
||||||
if (descriptor.resource_count < 2 or descriptor.resources[0].kind != @intFromEnum(device.ResourceKind.memory)) {
|
if (descriptor.resource_count < 2 or descriptor.resources[0].kind != @intFromEnum(device.ResourceKind.memory)) {
|
||||||
_ = runtime.system.write("pci-bus: bridge has no ECAM window\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: bridge has no ECAM window\n");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
const bus_range = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
const bus_range = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
||||||
if (resource.kind == @intFromEnum(device.ResourceKind.bus_range)) break resource;
|
if (resource.kind == @intFromEnum(device.ResourceKind.bus_range)) break resource;
|
||||||
} else {
|
} else {
|
||||||
_ = runtime.system.write("pci-bus: bridge has no bus range\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: bridge has no bus range\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
start_bus = bus_range.start;
|
start_bus = bus_range.start;
|
||||||
bus_count = bus_range.len;
|
bus_count = bus_range.len;
|
||||||
ecam_physical = descriptor.resources[0].start;
|
ecam_physical = descriptor.resources[0].start;
|
||||||
ecam_base = device.mmioMap(bridge_id, 0) orelse {
|
ecam_base = device.mmioMap(bridge_id, 0) orelse {
|
||||||
_ = runtime.system.write("pci-bus: ECAM mmio_map failed\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: ECAM mmio_map failed\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -116,17 +116,17 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
if (manager == null) runtime.system.sleep(20);
|
if (manager == null) runtime.system.sleep(20);
|
||||||
}
|
}
|
||||||
const h = manager orelse {
|
const h = manager orelse {
|
||||||
_ = runtime.system.write("pci-bus: no device manager to hello\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: no device manager to hello\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = bridge_id };
|
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = bridge_id };
|
||||||
var reply: [protocol.message_maximum]u8 = undefined;
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||||
_ = runtime.system.write("pci-bus: hello call failed\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: hello call failed\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||||
_ = runtime.system.write("pci-bus: hello refused\n");
|
_ = runtime.system.write("/system/drivers/pci-bus: hello refused\n");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
manager_handle = h;
|
manager_handle = h;
|
||||||
@@ -159,7 +159,7 @@ fn scan() void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
writeLine("pci-bus: {d} functions found\n", .{found});
|
writeLine("/system/drivers/pci-bus: {d} functions found\n", .{found});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Register one function under the bridge and report it to the manager. The
|
/// Register one function under the bridge and report it to the manager. The
|
||||||
@@ -229,7 +229,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
}
|
}
|
||||||
|
|
||||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||||
writeLine("pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
writeLine("/system/drivers/pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
const report = protocol.ChildAdded{
|
const report = protocol.ChildAdded{
|
||||||
@@ -240,7 +240,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
};
|
};
|
||||||
var reply: [protocol.message_maximum]u8 = undefined;
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||||
writeLine("pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
writeLine("/system/drivers/pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -255,7 +255,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
|||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||||
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
writeLine("pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
writeLine("/system/drivers/pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
runtime.service.run(protocol.message_maximum, .{
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
|||||||
@@ -0,0 +1,191 @@
|
|||||||
|
//! Pure decoders for USB HID **boot-protocol** reports — the simplified,
|
||||||
|
//! fixed-format reports a boot keyboard and boot mouse send, the USB analog of
|
||||||
|
//! the PS/2 scancode and mouse-packet decoders. No I/O: these turn report bytes
|
||||||
|
//! into make/break transitions and motion, which the usb-hid drivers publish to
|
||||||
|
//! the input service. Host-testable in isolation (like mouse-packet.zig).
|
||||||
|
//!
|
||||||
|
//! "Boot protocol" is a USB HID term (USB HID 1.11 §B) — the device reports in
|
||||||
|
//! this fixed layout after SET_PROTOCOL(boot); it has nothing to do with system
|
||||||
|
//! boot.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
// --- keyboard ---------------------------------------------------------------
|
||||||
|
|
||||||
|
/// The 8-byte boot keyboard report: a modifier bitmap, a reserved byte, and up
|
||||||
|
/// to six concurrently-pressed key usages.
|
||||||
|
pub const KeyboardReport = extern struct {
|
||||||
|
modifiers: u8 = 0,
|
||||||
|
reserved: u8 = 0,
|
||||||
|
keys: [6]u8 = .{ 0, 0, 0, 0, 0, 0 },
|
||||||
|
};
|
||||||
|
|
||||||
|
// The modifier byte's bits (HID keyboard boot report).
|
||||||
|
pub const modifier_left_control: u8 = 1 << 0;
|
||||||
|
pub const modifier_left_shift: u8 = 1 << 1;
|
||||||
|
pub const modifier_left_alt: u8 = 1 << 2;
|
||||||
|
pub const modifier_left_gui: u8 = 1 << 3;
|
||||||
|
pub const modifier_right_control: u8 = 1 << 4;
|
||||||
|
pub const modifier_right_shift: u8 = 1 << 5;
|
||||||
|
pub const modifier_right_alt: u8 = 1 << 6;
|
||||||
|
pub const modifier_right_gui: u8 = 1 << 7;
|
||||||
|
|
||||||
|
pub const TransitionKind = enum { pressed, released };
|
||||||
|
|
||||||
|
/// One key going down or up. `usage` is a HID keyboard-page usage — modifier keys
|
||||||
|
/// map to usages 224..231 — which is exactly the input protocol's `Keycode`.
|
||||||
|
pub const Transition = struct { kind: TransitionKind, usage: u8 };
|
||||||
|
|
||||||
|
// A report can change at most all 8 modifiers and all 6 keys at once.
|
||||||
|
pub const max_transitions = 8 + 6;
|
||||||
|
|
||||||
|
pub const Transitions = struct {
|
||||||
|
items: [max_transitions]Transition = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
|
||||||
|
fn add(self: *Transitions, transition: Transition) void {
|
||||||
|
if (self.count < self.items.len) {
|
||||||
|
self.items[self.count] = transition;
|
||||||
|
self.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn slice(self: *const Transitions) []const Transition {
|
||||||
|
return self.items[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Turns a stream of boot keyboard reports into make/break transitions by diffing
|
||||||
|
/// each report against the last.
|
||||||
|
pub const KeyboardDecoder = struct {
|
||||||
|
previous: KeyboardReport = .{},
|
||||||
|
|
||||||
|
pub fn feed(self: *KeyboardDecoder, current: KeyboardReport) Transitions {
|
||||||
|
var out = Transitions{};
|
||||||
|
|
||||||
|
// Rollover: 0x01 (ErrorRollOver) means more keys are held than the report
|
||||||
|
// can carry, so the key array is invalid. Emit nothing and keep the prior
|
||||||
|
// state (so the eventual releases still resolve against real keys).
|
||||||
|
for (current.keys) |key| {
|
||||||
|
if (key == 0x01) return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Modifiers: one make/break per changed bit; modifier usages are 224..231.
|
||||||
|
const changed = current.modifiers ^ self.previous.modifiers;
|
||||||
|
var bit: u3 = 0;
|
||||||
|
while (true) : (bit += 1) {
|
||||||
|
const mask = @as(u8, 1) << bit;
|
||||||
|
if (changed & mask != 0) {
|
||||||
|
out.add(.{
|
||||||
|
.kind = if (current.modifiers & mask != 0) .pressed else .released,
|
||||||
|
.usage = 224 + @as(u8, bit),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
if (bit == 7) break;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Keys made: present now, absent before.
|
||||||
|
for (current.keys) |key| {
|
||||||
|
if (key != 0 and !contains(&self.previous.keys, key)) out.add(.{ .kind = .pressed, .usage = key });
|
||||||
|
}
|
||||||
|
// Keys broken: present before, absent now.
|
||||||
|
for (self.previous.keys) |key| {
|
||||||
|
if (key != 0 and !contains(¤t.keys, key)) out.add(.{ .kind = .released, .usage = key });
|
||||||
|
}
|
||||||
|
|
||||||
|
self.previous = current;
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn contains(keys: *const [6]u8, value: u8) bool {
|
||||||
|
for (keys) |key| {
|
||||||
|
if (key == value) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- mouse ------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A decoded boot mouse report: the button bitmap and relative motion. The wheel
|
||||||
|
/// byte is present only on 4-byte reports (QEMU's usb-mouse sends one).
|
||||||
|
pub const MouseReport = struct {
|
||||||
|
buttons: u8 = 0,
|
||||||
|
dx: i8 = 0,
|
||||||
|
dy: i8 = 0,
|
||||||
|
wheel: i8 = 0,
|
||||||
|
has_wheel: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const mouse_button_left: u8 = 1 << 0;
|
||||||
|
pub const mouse_button_right: u8 = 1 << 1;
|
||||||
|
pub const mouse_button_middle: u8 = 1 << 2;
|
||||||
|
|
||||||
|
/// Parse a 3- or 4-byte boot mouse report. Note HID reports Y in screen
|
||||||
|
/// convention (positive = down), so — unlike PS/2 — `dy` is NOT negated.
|
||||||
|
pub fn parseMouse(bytes: []const u8) ?MouseReport {
|
||||||
|
if (bytes.len < 3) return null;
|
||||||
|
return .{
|
||||||
|
.buttons = bytes[0],
|
||||||
|
.dx = @bitCast(bytes[1]),
|
||||||
|
.dy = @bitCast(bytes[2]),
|
||||||
|
.wheel = if (bytes.len >= 4) @bitCast(bytes[3]) else 0,
|
||||||
|
.has_wheel = bytes.len >= 4,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
|
test "keyboard diff produces make and break transitions" {
|
||||||
|
var decoder = KeyboardDecoder{};
|
||||||
|
|
||||||
|
// Press 'a' (usage 4).
|
||||||
|
var t = decoder.feed(.{ .keys = .{ 4, 0, 0, 0, 0, 0 } });
|
||||||
|
try std.testing.expectEqual(@as(usize, 1), t.count);
|
||||||
|
try std.testing.expectEqual(TransitionKind.pressed, t.items[0].kind);
|
||||||
|
try std.testing.expectEqual(@as(u8, 4), t.items[0].usage);
|
||||||
|
|
||||||
|
// Hold 'a', press 'b' (usage 5): only 'b' is new.
|
||||||
|
t = decoder.feed(.{ .keys = .{ 4, 5, 0, 0, 0, 0 } });
|
||||||
|
try std.testing.expectEqual(@as(usize, 1), t.count);
|
||||||
|
try std.testing.expectEqual(@as(u8, 5), t.items[0].usage);
|
||||||
|
|
||||||
|
// Release everything: 'a' and 'b' both break.
|
||||||
|
t = decoder.feed(.{ .keys = .{ 0, 0, 0, 0, 0, 0 } });
|
||||||
|
try std.testing.expectEqual(@as(usize, 2), t.count);
|
||||||
|
try std.testing.expectEqual(TransitionKind.released, t.items[0].kind);
|
||||||
|
|
||||||
|
// Press Left Shift (modifier bit 1 -> usage 225).
|
||||||
|
t = decoder.feed(.{ .modifiers = modifier_left_shift });
|
||||||
|
try std.testing.expectEqual(@as(usize, 1), t.count);
|
||||||
|
try std.testing.expectEqual(@as(u8, 225), t.items[0].usage);
|
||||||
|
try std.testing.expectEqual(TransitionKind.pressed, t.items[0].kind);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "rollover report is ignored but state is preserved" {
|
||||||
|
var decoder = KeyboardDecoder{};
|
||||||
|
_ = decoder.feed(.{ .keys = .{ 4, 0, 0, 0, 0, 0 } }); // press 'a'
|
||||||
|
|
||||||
|
const rollover = decoder.feed(.{ .keys = .{ 0x01, 0x01, 0x01, 0x01, 0x01, 0x01 } });
|
||||||
|
try std.testing.expectEqual(@as(usize, 0), rollover.count);
|
||||||
|
|
||||||
|
// 'a' is still considered down, so releasing all keys now breaks it.
|
||||||
|
const release = decoder.feed(.{ .keys = .{ 0, 0, 0, 0, 0, 0 } });
|
||||||
|
try std.testing.expectEqual(@as(usize, 1), release.count);
|
||||||
|
try std.testing.expectEqual(@as(u8, 4), release.items[0].usage);
|
||||||
|
try std.testing.expectEqual(TransitionKind.released, release.items[0].kind);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "mouse report parses motion without inverting Y" {
|
||||||
|
const three = parseMouse(&.{ mouse_button_left, 5, 0xFB }).?; // dy = -5
|
||||||
|
try std.testing.expectEqual(mouse_button_left, three.buttons);
|
||||||
|
try std.testing.expectEqual(@as(i8, 5), three.dx);
|
||||||
|
try std.testing.expectEqual(@as(i8, -5), three.dy);
|
||||||
|
try std.testing.expect(!three.has_wheel);
|
||||||
|
|
||||||
|
const four = parseMouse(&.{ 0, 0, 0, 0xFF }).?; // wheel = -1
|
||||||
|
try std.testing.expect(four.has_wheel);
|
||||||
|
try std.testing.expectEqual(@as(i8, -1), four.wheel);
|
||||||
|
|
||||||
|
try std.testing.expect(parseMouse(&.{ 0, 0 }) == null); // too short
|
||||||
|
}
|
||||||
@@ -0,0 +1,174 @@
|
|||||||
|
//! USB HID boot keyboard driver.
|
||||||
|
//!
|
||||||
|
//! Spawned by the device manager when the xHCI bus driver reports a HID / boot /
|
||||||
|
//! keyboard interface (class 3, subclass 1, protocol 1); its assigned device id
|
||||||
|
//! arrives as argv[1] and an optional layout name ("us", "gb", ...) as argv[2].
|
||||||
|
//! It owns no hardware: it opens its device through the USB transfer protocol
|
||||||
|
//! (`runtime.usb`), asks the device for the boot protocol, subscribes to its
|
||||||
|
//! interrupt-IN endpoint, and turns each 8-byte boot report into input-protocol
|
||||||
|
//! events, published to the input service — the USB analogue of ps2-bus/keyboard.
|
||||||
|
//!
|
||||||
|
//! interrupt report -> hid-report diff -> key_down / key_up
|
||||||
|
//! -> xkeyboard-config -> character -> key_press
|
||||||
|
//!
|
||||||
|
//! Because a USB keyboard's usages ARE the input protocol's keycodes (both are
|
||||||
|
//! HID keyboard page 0x07), the decode is nearly 1:1 — no scancode translation.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const usb_abi = @import("usb-abi");
|
||||||
|
const xkb = @import("xkeyboard-config");
|
||||||
|
const hid = @import("hid-report.zig");
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const process = runtime.process;
|
||||||
|
const input_protocol = runtime.input_protocol;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The modifier state a character lookup needs — derived from the report's
|
||||||
|
// modifier byte, plus the driver-tracked caps-lock toggle.
|
||||||
|
const ModifierSnapshot = struct {
|
||||||
|
shift: bool,
|
||||||
|
control: bool,
|
||||||
|
right_alt: bool,
|
||||||
|
caps_lock: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The character a key produces under `modifiers`, or 0 for none — the layout
|
||||||
|
/// lookup for printable keys, with ASCII control characters for the keys every
|
||||||
|
/// consumer expects (Enter, Tab, Backspace, Escape), exactly as ps2-bus/keyboard.
|
||||||
|
fn characterFor(layout: *const xkb.Layout, usage: u8, modifiers: ModifierSnapshot) u32 {
|
||||||
|
const mapping = xkb.map(layout, usage, .{
|
||||||
|
.shift = modifiers.shift,
|
||||||
|
.caps_lock = modifiers.caps_lock,
|
||||||
|
.level3 = modifiers.right_alt,
|
||||||
|
.control = modifiers.control,
|
||||||
|
});
|
||||||
|
if (mapping.character) |character| return character;
|
||||||
|
return switch (@as(input_protocol.Keycode, @enumFromInt(usage))) {
|
||||||
|
.enter, .keypad_enter => '\n',
|
||||||
|
.tab => '\t',
|
||||||
|
.backspace => 0x08,
|
||||||
|
.escape => 0x1B,
|
||||||
|
else => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn modifierWord(modifiers: u8) u32 {
|
||||||
|
var word: u32 = 0;
|
||||||
|
if (modifiers & (hid.modifier_left_shift | hid.modifier_right_shift) != 0) word |= input_protocol.modifier_shift;
|
||||||
|
if (modifiers & (hid.modifier_left_control | hid.modifier_right_control) != 0) word |= input_protocol.modifier_control;
|
||||||
|
if (modifiers & (hid.modifier_left_alt | hid.modifier_right_alt) != 0) word |= input_protocol.modifier_alt;
|
||||||
|
return word;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: missing device id (argv[1])\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
writeLine("/system/drivers/usb-hid/keyboard: malformed device id '{s}'\n", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const layout = xkb.byName(init.arguments.get(2) orelse "us") orelse xkb.us;
|
||||||
|
|
||||||
|
// Hello the manager first (meet the spawn deadline), then open the device.
|
||||||
|
if (!runtime.usb.helloManager(device_id)) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: hello to device manager failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
var device = runtime.usb.open(device_id) orelse {
|
||||||
|
writeLine("/system/drivers/usb-hid/keyboard: could not open device {d}\n", .{device_id});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: no interrupt-IN endpoint\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Ask for the boot protocol and an indefinite idle (report only on change).
|
||||||
|
_ = device.controlOut(@bitCast(usb_abi.setProtocol(@enumFromInt(device.interface_number), .boot)));
|
||||||
|
_ = device.controlOut(@bitCast(usb_abi.setIdle(@enumFromInt(device.interface_number), 0, 0)));
|
||||||
|
|
||||||
|
if (!device.subscribeInterrupt(endpoint.address, endpoint.max_packet_size)) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: interrupt subscribe failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var source = runtime.input.connectSource() orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: input service unavailable\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
_ = process.bindSignals(device.endpoint);
|
||||||
|
writeLine("/system/drivers/usb-hid/keyboard: ok (device {d}, interface {d}, layout {s})\n", .{ device_id, device.interface_number, layout.name });
|
||||||
|
|
||||||
|
var decoder = hid.KeyboardDecoder{};
|
||||||
|
var caps_lock = false;
|
||||||
|
var receive: [64]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(device.endpoint, &.{}, &receive, null);
|
||||||
|
if (!got.isNotification()) continue;
|
||||||
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
|
if (signals.has(.terminate)) return;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!got.isMessage() or got.len < @sizeOf(runtime.usb.InterruptReport)) continue;
|
||||||
|
|
||||||
|
const message = std.mem.bytesToValue(runtime.usb.InterruptReport, receive[0..@sizeOf(runtime.usb.InterruptReport)]);
|
||||||
|
if (message.length < @sizeOf(hid.KeyboardReport)) continue;
|
||||||
|
const report = std.mem.bytesToValue(hid.KeyboardReport, message.data[0..@sizeOf(hid.KeyboardReport)]);
|
||||||
|
const transitions = decoder.feed(report);
|
||||||
|
|
||||||
|
// Caps Lock toggles on its own key-down (a stateful lock, not a modifier).
|
||||||
|
for (transitions.slice()) |transition| {
|
||||||
|
if (transition.kind == .pressed and @as(input_protocol.Keycode, @enumFromInt(transition.usage)) == .caps_lock) caps_lock = !caps_lock;
|
||||||
|
}
|
||||||
|
|
||||||
|
const modifiers = ModifierSnapshot{
|
||||||
|
.shift = report.modifiers & (hid.modifier_left_shift | hid.modifier_right_shift) != 0,
|
||||||
|
.control = report.modifiers & (hid.modifier_left_control | hid.modifier_right_control) != 0,
|
||||||
|
.right_alt = report.modifiers & hid.modifier_right_alt != 0,
|
||||||
|
.caps_lock = caps_lock,
|
||||||
|
};
|
||||||
|
const modifier_word = modifierWord(report.modifiers);
|
||||||
|
|
||||||
|
for (transitions.slice()) |transition| {
|
||||||
|
switch (transition.kind) {
|
||||||
|
.pressed => {
|
||||||
|
_ = source.publishKeyboardEvent(.{
|
||||||
|
.kind = @intFromEnum(input_protocol.EventKind.key_down),
|
||||||
|
.keycode = transition.usage,
|
||||||
|
.character = 0,
|
||||||
|
.modifiers = modifier_word,
|
||||||
|
});
|
||||||
|
const character = characterFor(layout, transition.usage, modifiers);
|
||||||
|
if (character != 0) {
|
||||||
|
_ = source.publishKeyboardEvent(.{
|
||||||
|
.kind = @intFromEnum(input_protocol.EventKind.key_press),
|
||||||
|
.keycode = transition.usage,
|
||||||
|
.character = character,
|
||||||
|
.modifiers = modifier_word,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
},
|
||||||
|
.released => {
|
||||||
|
_ = source.publishKeyboardEvent(.{
|
||||||
|
.kind = @intFromEnum(input_protocol.EventKind.key_up),
|
||||||
|
.keycode = transition.usage,
|
||||||
|
.character = 0,
|
||||||
|
.modifiers = modifier_word,
|
||||||
|
});
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,140 @@
|
|||||||
|
//! USB HID boot mouse driver.
|
||||||
|
//!
|
||||||
|
//! Spawned by the device manager when the xHCI bus driver reports a HID / boot /
|
||||||
|
//! mouse interface (class 3, subclass 1, protocol 2); its assigned device id
|
||||||
|
//! arrives as argv[1]. Like the keyboard driver it owns no hardware: it opens its
|
||||||
|
//! device through the USB transfer protocol (`runtime.usb`), asks for the boot
|
||||||
|
//! protocol, subscribes to its interrupt-IN endpoint, and turns each 3- or 4-byte
|
||||||
|
//! boot report into input-protocol mouse events published to the input service.
|
||||||
|
//!
|
||||||
|
//! Unlike PS/2, HID reports Y in screen convention (positive = down), so motion
|
||||||
|
//! is passed straight through (the decode in hid-report.zig does not negate it).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const usb_abi = @import("usb-abi");
|
||||||
|
const hid = @import("hid-report.zig");
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const process = runtime.process;
|
||||||
|
const input_protocol = runtime.input_protocol;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The current pressed-button bitmask in input-protocol terms.
|
||||||
|
fn buttonMask(buttons: u8) u32 {
|
||||||
|
var mask: u32 = 0;
|
||||||
|
if (buttons & hid.mouse_button_left != 0) mask |= input_protocol.mouse_button_left;
|
||||||
|
if (buttons & hid.mouse_button_right != 0) mask |= input_protocol.mouse_button_right;
|
||||||
|
if (buttons & hid.mouse_button_middle != 0) mask |= input_protocol.mouse_button_middle;
|
||||||
|
return mask;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/mouse: missing device id (argv[1])\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
writeLine("/system/drivers/usb-hid/mouse: malformed device id '{s}'\n", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!runtime.usb.helloManager(device_id)) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/mouse: hello to device manager failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
var device = runtime.usb.open(device_id) orelse {
|
||||||
|
writeLine("/system/drivers/usb-hid/mouse: could not open device {d}\n", .{device_id});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/mouse: no interrupt-IN endpoint\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
_ = device.controlOut(@bitCast(usb_abi.setProtocol(@enumFromInt(device.interface_number), .boot)));
|
||||||
|
|
||||||
|
if (!device.subscribeInterrupt(endpoint.address, endpoint.max_packet_size)) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/mouse: interrupt subscribe failed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var source = runtime.input.connectSource() orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-hid/mouse: input service unavailable\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
_ = process.bindSignals(device.endpoint);
|
||||||
|
writeLine("/system/drivers/usb-hid/mouse: ok (device {d}, interface {d})\n", .{ device_id, device.interface_number });
|
||||||
|
|
||||||
|
var previous_buttons: u8 = 0;
|
||||||
|
var receive: [64]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(device.endpoint, &.{}, &receive, null);
|
||||||
|
if (!got.isNotification()) continue;
|
||||||
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
|
if (signals.has(.terminate)) return;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!got.isMessage() or got.len < @sizeOf(runtime.usb.InterruptReport)) continue;
|
||||||
|
|
||||||
|
const message = std.mem.bytesToValue(runtime.usb.InterruptReport, receive[0..@sizeOf(runtime.usb.InterruptReport)]);
|
||||||
|
const length = @min(message.length, message.data.len);
|
||||||
|
const report = hid.parseMouse(message.data[0..length]) orelse continue;
|
||||||
|
const mask = buttonMask(report.buttons);
|
||||||
|
|
||||||
|
// Button transitions: one event per changed button bit.
|
||||||
|
const changed = report.buttons ^ previous_buttons;
|
||||||
|
inline for (.{
|
||||||
|
.{ hid.mouse_button_left, input_protocol.mouse_button_left },
|
||||||
|
.{ hid.mouse_button_right, input_protocol.mouse_button_right },
|
||||||
|
.{ hid.mouse_button_middle, input_protocol.mouse_button_middle },
|
||||||
|
}) |pair| {
|
||||||
|
if (changed & pair[0] != 0) {
|
||||||
|
_ = source.publishMouseEvent(.{
|
||||||
|
.kind = @intFromEnum(if (report.buttons & pair[0] != 0) input_protocol.MouseEventKind.button_down else input_protocol.MouseEventKind.button_up),
|
||||||
|
.button = pair[1],
|
||||||
|
.dx = 0,
|
||||||
|
.dy = 0,
|
||||||
|
.scroll_x = 0,
|
||||||
|
.scroll_y = 0,
|
||||||
|
.buttons = mask,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
previous_buttons = report.buttons;
|
||||||
|
|
||||||
|
// Relative motion (dy straight through — HID Y is already screen convention).
|
||||||
|
if (report.dx != 0 or report.dy != 0) {
|
||||||
|
_ = source.publishMouseEvent(.{
|
||||||
|
.kind = @intFromEnum(input_protocol.MouseEventKind.motion),
|
||||||
|
.button = 0,
|
||||||
|
.dx = report.dx,
|
||||||
|
.dy = report.dy,
|
||||||
|
.scroll_x = 0,
|
||||||
|
.scroll_y = 0,
|
||||||
|
.buttons = mask,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Wheel (4-byte reports only): positive = scroll up.
|
||||||
|
if (report.has_wheel and report.wheel != 0) {
|
||||||
|
_ = source.publishMouseEvent(.{
|
||||||
|
.kind = @intFromEnum(input_protocol.MouseEventKind.scroll),
|
||||||
|
.button = 0,
|
||||||
|
.dx = 0,
|
||||||
|
.dy = 0,
|
||||||
|
.scroll_x = 0,
|
||||||
|
.scroll_y = report.wheel,
|
||||||
|
.buttons = mask,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
//! USB Mass Storage Bulk-Only Transport (BOT) wire structures — the Command and
|
||||||
|
//! Command Status Wrappers that bracket every command (USB MSC BOT §5). Pure data
|
||||||
|
//! definitions, host-testable in isolation. The command inside the CBW is a SCSI
|
||||||
|
//! CDB (see scsi.zig); the transport here just carries it and reports status.
|
||||||
|
//!
|
||||||
|
//! One command is three bulk transfers: CBW out, an optional data stage, CSW in.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// "USBC" — the signature at the head of every Command Block Wrapper.
|
||||||
|
pub const cbw_signature: u32 = 0x43425355;
|
||||||
|
/// "USBS" — the signature at the head of every Command Status Wrapper.
|
||||||
|
pub const csw_signature: u32 = 0x53425355;
|
||||||
|
|
||||||
|
/// CBW `flags`: set for a device-to-host (IN) data stage, clear for OUT.
|
||||||
|
pub const flag_data_in: u8 = 0x80;
|
||||||
|
|
||||||
|
/// The 31-byte Command Block Wrapper, sent on the bulk-OUT endpoint.
|
||||||
|
pub const CommandBlockWrapper = extern struct {
|
||||||
|
signature: u32 align(1) = cbw_signature,
|
||||||
|
tag: u32 align(1),
|
||||||
|
data_transfer_length: u32 align(1),
|
||||||
|
flags: u8,
|
||||||
|
lun: u8,
|
||||||
|
cdb_length: u8,
|
||||||
|
cdb: [16]u8 = [_]u8{0} ** 16,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A device's answer to a command (the CSW `status` byte).
|
||||||
|
pub const CommandStatus = enum(u8) {
|
||||||
|
passed = 0,
|
||||||
|
failed = 1,
|
||||||
|
phase_error = 2,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The 13-byte Command Status Wrapper, read from the bulk-IN endpoint.
|
||||||
|
pub const CommandStatusWrapper = extern struct {
|
||||||
|
signature: u32 align(1) = csw_signature,
|
||||||
|
tag: u32 align(1),
|
||||||
|
data_residue: u32 align(1),
|
||||||
|
status: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
comptime {
|
||||||
|
std.debug.assert(@sizeOf(CommandBlockWrapper) == 31);
|
||||||
|
std.debug.assert(@sizeOf(CommandStatusWrapper) == 13);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "wrapper sizes and signatures match the specification" {
|
||||||
|
const cbw = CommandBlockWrapper{
|
||||||
|
.tag = 0x11223344,
|
||||||
|
.data_transfer_length = 512,
|
||||||
|
.flags = flag_data_in,
|
||||||
|
.lun = 0,
|
||||||
|
.cdb_length = 10,
|
||||||
|
};
|
||||||
|
const bytes = std.mem.asBytes(&cbw);
|
||||||
|
try std.testing.expectEqual(@as(usize, 31), bytes.len);
|
||||||
|
// "USBC" little-endian.
|
||||||
|
try std.testing.expectEqualSlices(u8, "USBC", bytes[0..4]);
|
||||||
|
try std.testing.expectEqual(flag_data_in, bytes[12]);
|
||||||
|
|
||||||
|
const csw = std.mem.bytesToValue(CommandStatusWrapper, &[_]u8{
|
||||||
|
0x55, 0x53, 0x42, 0x53, // "USBS"
|
||||||
|
0x44, 0x33, 0x22, 0x11, // tag
|
||||||
|
0x00, 0x00, 0x00, 0x00, // residue
|
||||||
|
0x00, // passed
|
||||||
|
});
|
||||||
|
try std.testing.expectEqual(csw_signature, csw.signature);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x11223344), csw.tag);
|
||||||
|
try std.testing.expectEqual(@as(u8, @intFromEnum(CommandStatus.passed)), csw.status);
|
||||||
|
}
|
||||||
@@ -0,0 +1,97 @@
|
|||||||
|
//! The SCSI command descriptor blocks a transparent-SCSI (subclass 0x06) mass
|
||||||
|
//! storage device understands, and the parsers for what they return. Pure data —
|
||||||
|
//! host-testable. These CDBs go inside a Bulk-Only-Transport CBW (see
|
||||||
|
//! bulk-only-transport.zig).
|
||||||
|
//!
|
||||||
|
//! Every multi-byte SCSI field is **big-endian** — the opposite of the USB wire
|
||||||
|
//! ABI — so the LBA and transfer-length encodings are the load-bearing detail.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
// SCSI operation codes.
|
||||||
|
const op_test_unit_ready: u8 = 0x00;
|
||||||
|
const op_request_sense: u8 = 0x03;
|
||||||
|
const op_inquiry: u8 = 0x12;
|
||||||
|
const op_read_capacity_10: u8 = 0x25;
|
||||||
|
const op_read_10: u8 = 0x28;
|
||||||
|
const op_write_10: u8 = 0x2A;
|
||||||
|
const op_synchronize_cache_10: u8 = 0x35;
|
||||||
|
|
||||||
|
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
||||||
|
/// and product strings).
|
||||||
|
pub fn inquiry(allocation_length: u8) [6]u8 {
|
||||||
|
return .{ op_inquiry, 0, 0, 0, allocation_length, 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// TEST UNIT READY: no data; success (CSW passed) means the unit is ready.
|
||||||
|
pub fn testUnitReady() [6]u8 {
|
||||||
|
return .{ op_test_unit_ready, 0, 0, 0, 0, 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// REQUEST SENSE: 18 bytes of sense data (sense key + ASC/ASCQ) explaining the
|
||||||
|
/// previous failure.
|
||||||
|
pub fn requestSense(allocation_length: u8) [6]u8 {
|
||||||
|
return .{ op_request_sense, 0, 0, 0, allocation_length, 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// READ CAPACITY(10): 8 bytes back — the last LBA and the block size, both u32
|
||||||
|
/// big-endian. Block count is last_lba + 1.
|
||||||
|
pub fn readCapacity10() [10]u8 {
|
||||||
|
return .{ op_read_capacity_10, 0, 0, 0, 0, 0, 0, 0, 0, 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// READ(10): read `blocks` logical blocks starting at `lba` into the data stage.
|
||||||
|
pub fn read10(lba: u32, blocks: u16) [10]u8 {
|
||||||
|
var cdb = [_]u8{0} ** 10;
|
||||||
|
cdb[0] = op_read_10;
|
||||||
|
std.mem.writeInt(u32, cdb[2..6], lba, .big);
|
||||||
|
std.mem.writeInt(u16, cdb[7..9], blocks, .big);
|
||||||
|
return cdb;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// WRITE(10): write `blocks` logical blocks starting at `lba` from the data stage.
|
||||||
|
pub fn write10(lba: u32, blocks: u16) [10]u8 {
|
||||||
|
var cdb = [_]u8{0} ** 10;
|
||||||
|
cdb[0] = op_write_10;
|
||||||
|
std.mem.writeInt(u32, cdb[2..6], lba, .big);
|
||||||
|
std.mem.writeInt(u16, cdb[7..9], blocks, .big);
|
||||||
|
return cdb;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SYNCHRONIZE CACHE(10): commit the device's write cache to stable media. LBA 0
|
||||||
|
/// and block count 0 mean "the whole medium". No data stage. Without this a write
|
||||||
|
/// can sit in the USB flash controller's cache and be lost if power is cut right
|
||||||
|
/// after — which is exactly what a shutdown-time log flush hits on real hardware.
|
||||||
|
pub fn synchronizeCache10() [10]u8 {
|
||||||
|
var cdb = [_]u8{0} ** 10;
|
||||||
|
cdb[0] = op_synchronize_cache_10;
|
||||||
|
return cdb;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode an 8-byte READ CAPACITY(10) reply.
|
||||||
|
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
||||||
|
return .{
|
||||||
|
.last_lba = std.mem.readInt(u32, bytes[0..4], .big),
|
||||||
|
.block_size = std.mem.readInt(u32, bytes[4..8], .big),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "read/write CDBs encode the LBA and length big-endian" {
|
||||||
|
const read = read10(0x01020304, 8);
|
||||||
|
try std.testing.expectEqualSlices(u8, &.{ 0x28, 0x00, 0x01, 0x02, 0x03, 0x04, 0x00, 0x00, 0x08, 0x00 }, &read);
|
||||||
|
|
||||||
|
const write = write10(0xAABBCCDD, 1);
|
||||||
|
try std.testing.expectEqualSlices(u8, &.{ 0x2A, 0x00, 0xAA, 0xBB, 0xCC, 0xDD, 0x00, 0x00, 0x01, 0x00 }, &write);
|
||||||
|
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x25), readCapacity10()[0]);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x12), inquiry(36)[0]);
|
||||||
|
try std.testing.expectEqual(@as(u8, 36), inquiry(36)[4]);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x00), testUnitReady()[0]);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "read capacity parses last LBA and block size" {
|
||||||
|
// last_lba = 0x0003FFFF (262144 blocks), block_size = 512.
|
||||||
|
const capacity = parseCapacity(.{ 0x00, 0x03, 0xFF, 0xFF, 0x00, 0x00, 0x02, 0x00 });
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0003FFFF), capacity.last_lba);
|
||||||
|
try std.testing.expectEqual(@as(u32, 512), capacity.block_size);
|
||||||
|
}
|
||||||
@@ -0,0 +1,187 @@
|
|||||||
|
//! USB mass-storage class driver (Bulk-Only Transport + transparent SCSI).
|
||||||
|
//!
|
||||||
|
//! Spawned by the device manager when the xHCI bus driver reports a mass-storage
|
||||||
|
//! / SCSI / bulk-only interface (class 8, subclass 6, protocol 0x50); its device
|
||||||
|
//! id arrives as argv[1]. It owns no hardware: it opens its device through the
|
||||||
|
//! USB transfer protocol (`runtime.usb`), then drives it with the BOT command
|
||||||
|
//! cycle — CBW out, an optional data stage, CSW in — carrying SCSI commands
|
||||||
|
//! (READ CAPACITY, READ(10), WRITE(10)). Upward it is a block device: it serves
|
||||||
|
//! the block protocol under `.block`, the storage a FAT filesystem sits on.
|
||||||
|
//!
|
||||||
|
//! Block data never crosses IPC: read/write name a caller-owned DMA buffer by
|
||||||
|
//! physical address, which the data stage DMAs straight to/from.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const scsi = @import("scsi.zig");
|
||||||
|
const bot = @import("bulk-only-transport.zig");
|
||||||
|
const block_protocol = @import("block-protocol");
|
||||||
|
const dma = runtime.dma;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
var device_id: u64 = 0;
|
||||||
|
var device: runtime.usb.Device = undefined;
|
||||||
|
var bulk_in: runtime.usb.Endpoint = undefined;
|
||||||
|
var bulk_out: runtime.usb.Endpoint = undefined;
|
||||||
|
|
||||||
|
// DMA buffers for the transport: the 31-byte CBW, the 13-byte CSW, and a page
|
||||||
|
// for the small command data (INQUIRY / READ CAPACITY / the self-check sector).
|
||||||
|
var command_wrapper: dma.Region = undefined;
|
||||||
|
var status_wrapper: dma.Region = undefined;
|
||||||
|
var command_data: dma.Region = undefined;
|
||||||
|
|
||||||
|
var next_tag: u32 = 1;
|
||||||
|
var block_size: u32 = 512;
|
||||||
|
var block_count: u64 = 0;
|
||||||
|
|
||||||
|
/// One Bulk-Only-Transport command: send the CBW, run the data stage (to/from
|
||||||
|
/// `data_physical`), read and validate the CSW. Returns true on a passed status.
|
||||||
|
fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length: u32) bool {
|
||||||
|
const tag = next_tag;
|
||||||
|
next_tag +%= 1;
|
||||||
|
|
||||||
|
const wrapper: *bot.CommandBlockWrapper = @ptrFromInt(command_wrapper.virtual);
|
||||||
|
wrapper.* = .{
|
||||||
|
.tag = tag,
|
||||||
|
.data_transfer_length = data_length,
|
||||||
|
.flags = if (direction_in) bot.flag_data_in else 0,
|
||||||
|
.lun = 0,
|
||||||
|
.cdb_length = @intCast(cdb.len),
|
||||||
|
};
|
||||||
|
@memcpy(wrapper.cdb[0..cdb.len], cdb);
|
||||||
|
|
||||||
|
if (device.bulk(bulk_out.address, command_wrapper.physical, @sizeOf(bot.CommandBlockWrapper)) == null) return false;
|
||||||
|
if (data_length > 0) {
|
||||||
|
const endpoint = if (direction_in) bulk_in.address else bulk_out.address;
|
||||||
|
if (device.bulk(endpoint, data_physical, data_length) == null) return false;
|
||||||
|
}
|
||||||
|
if (device.bulk(bulk_in.address, status_wrapper.physical, @sizeOf(bot.CommandStatusWrapper)) == null) return false;
|
||||||
|
|
||||||
|
const status: *const bot.CommandStatusWrapper = @ptrFromInt(status_wrapper.virtual);
|
||||||
|
if (status.signature != bot.csw_signature or status.tag != tag) return false;
|
||||||
|
return status.status == @intFromEnum(bot.CommandStatus.passed);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
|
_ = endpoint;
|
||||||
|
if (!runtime.usb.helloManager(device_id)) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-storage: hello to device manager failed\n");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
device = runtime.usb.open(device_id) orelse {
|
||||||
|
writeLine("/system/drivers/usb-storage: could not open device {d}\n", .{device_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
bulk_in = device.findEndpoint(runtime.usb.transfer_type_bulk, true) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-storage: no bulk-IN endpoint\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
bulk_out = device.findEndpoint(runtime.usb.transfer_type_bulk, false) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-storage: no bulk-OUT endpoint\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
command_wrapper = dma.alloc(4096, dma.coherent) orelse return false;
|
||||||
|
status_wrapper = dma.alloc(4096, dma.coherent) orelse return false;
|
||||||
|
command_data = dma.alloc(4096, dma.coherent) orelse return false;
|
||||||
|
|
||||||
|
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
||||||
|
// with REQUEST SENSE), identify it, and read its capacity.
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (tries < 10) : (tries += 1) {
|
||||||
|
const ready = scsi.testUnitReady();
|
||||||
|
if (transact(&ready, false, 0, 0)) break;
|
||||||
|
const sense = scsi.requestSense(18);
|
||||||
|
_ = transact(&sense, true, command_data.physical, 18);
|
||||||
|
runtime.system.sleep(50);
|
||||||
|
}
|
||||||
|
const inquiry = scsi.inquiry(36);
|
||||||
|
_ = transact(&inquiry, true, command_data.physical, 36);
|
||||||
|
|
||||||
|
const capacity_command = scsi.readCapacity10();
|
||||||
|
if (!transact(&capacity_command, true, command_data.physical, 8)) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-storage: READ CAPACITY failed\n");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
var capacity_bytes: [8]u8 = undefined;
|
||||||
|
const capacity_source: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||||
|
@memcpy(&capacity_bytes, capacity_source[0..8]);
|
||||||
|
const capacity = scsi.parseCapacity(capacity_bytes);
|
||||||
|
block_size = capacity.block_size;
|
||||||
|
block_count = @as(u64, capacity.last_lba) + 1;
|
||||||
|
writeLine("/system/drivers/usb-storage: ready ({d} blocks x {d} bytes)\n", .{ block_count, block_size });
|
||||||
|
|
||||||
|
// Self-check: read block 0 and log its trailing signature (0x55AA for a boot
|
||||||
|
// sector) — proof READ(10) works end to end over the bulk path.
|
||||||
|
const read0 = scsi.read10(0, 1);
|
||||||
|
if (block_size <= 4096 and transact(&read0, true, command_data.physical, block_size)) {
|
||||||
|
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||||
|
writeLine("/system/drivers/usb-storage: block 0 signature 0x{x:0>2}{x:0>2}\n", .{ sector[510], sector[511] });
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serve the block protocol: geometry, and whole-block read/write to/from the
|
||||||
|
/// caller's DMA buffer (named by physical address).
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
if (message.len < block_protocol.request_size) return 0;
|
||||||
|
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||||
|
switch (request.operation) {
|
||||||
|
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||||
|
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||||
|
},
|
||||||
|
@intFromEnum(block_protocol.Operation.read) => {
|
||||||
|
const count: u16 = @intCast(request.count);
|
||||||
|
const cdb = scsi.read10(@intCast(request.lba), count);
|
||||||
|
const ok = transact(&cdb, true, request.physical, request.count * block_size);
|
||||||
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||||
|
},
|
||||||
|
@intFromEnum(block_protocol.Operation.write) => {
|
||||||
|
const count: u16 = @intCast(request.count);
|
||||||
|
const cdb = scsi.write10(@intCast(request.lba), count);
|
||||||
|
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||||
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||||
|
},
|
||||||
|
@intFromEnum(block_protocol.Operation.flush) => {
|
||||||
|
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||||
|
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||||
|
// cuts power. A device without a volatile cache reports success anyway.
|
||||||
|
const cdb = scsi.synchronizeCache10();
|
||||||
|
const ok = transact(&cdb, false, 0, 0);
|
||||||
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||||
|
},
|
||||||
|
else => return 0,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeReply(reply: []u8, value: block_protocol.Reply) usize {
|
||||||
|
const bytes = std.mem.asBytes(&value);
|
||||||
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
|
return bytes.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-storage: missing device id (argv[1])\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
writeLine("/system/drivers/usb-storage: malformed device id '{s}'\n", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
runtime.service.run(block_protocol.message_maximum, .{
|
||||||
|
.service = .block,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
//! The USB transfer protocol: what a USB class driver (a keyboard, mouse, or
|
||||||
|
//! mass-storage driver) says to the xHCI bus driver over its well-known
|
||||||
|
//! `.usb_bus` endpoint to drive its device. The class driver owns no hardware —
|
||||||
|
//! it reaches its device entirely through these messages, the way a PS/2 keyboard
|
||||||
|
//! driver reaches the 8042 through the ps2-bus. Extern-struct messages tagged by
|
||||||
|
//! `Operation`, the vfs-protocol / device-manager-protocol pattern.
|
||||||
|
//!
|
||||||
|
//! The shape:
|
||||||
|
//! - **open** (a capability-passing `ipc.callCap`): the class driver hands over
|
||||||
|
//! its own endpoint (for asynchronous interrupt reports) and its assigned
|
||||||
|
//! device id, and receives a `device_token` plus its interface's endpoints.
|
||||||
|
//! - **control / bulk** (synchronous `ipc.call`): one transfer, answered when
|
||||||
|
//! it completes. Control data travels inline (descriptors, HID/MSC class
|
||||||
|
//! requests are all small); bulk data travels by **physical address** — the
|
||||||
|
//! class driver's own `dma_alloc`'d buffer — so a 512-byte sector never has
|
||||||
|
//! to cross the 256-byte IPC boundary.
|
||||||
|
//! - **interrupt_subscribe** (synchronous): arm periodic IN polling of an
|
||||||
|
//! interrupt endpoint; each report the device produces is then pushed to the
|
||||||
|
//! class driver's endpoint as an asynchronous `InterruptReport` (`ipc.send`),
|
||||||
|
//! exactly how the input service delivers events.
|
||||||
|
//!
|
||||||
|
//! Single controller assumption: one `.usb_bus` singleton serves QEMU's one xHCI.
|
||||||
|
//! A multi-controller machine would need a per-controller endpoint (the device
|
||||||
|
//! manager handing each class driver the right one); noted, not built.
|
||||||
|
|
||||||
|
/// Fits one synchronous IPC message (kernel MESSAGE_MAXIMUM).
|
||||||
|
pub const message_maximum: usize = 256;
|
||||||
|
|
||||||
|
/// The largest inline control-transfer payload. Sized so a whole message
|
||||||
|
/// (header + data) stays under `message_maximum`: descriptors and HID/MSC class
|
||||||
|
/// requests are all far smaller.
|
||||||
|
pub const max_inline_data: usize = 200;
|
||||||
|
|
||||||
|
/// The largest interrupt report pushed asynchronously. Sized so `InterruptReport`
|
||||||
|
/// fits an `ipc_send` payload slot (POST_MAXIMUM = 64): boot keyboard reports are
|
||||||
|
/// 8 bytes, boot mouse reports 3–4.
|
||||||
|
pub const max_report_data: usize = 48;
|
||||||
|
|
||||||
|
/// Endpoints per interface reported back in an open reply (a boot HID interface
|
||||||
|
/// has one interrupt endpoint, a mass-storage interface two bulk endpoints).
|
||||||
|
pub const max_reported_endpoints: usize = 4;
|
||||||
|
|
||||||
|
pub const Operation = enum(u32) {
|
||||||
|
open = 0,
|
||||||
|
control = 1,
|
||||||
|
interrupt_subscribe = 2,
|
||||||
|
bulk = 3,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||||
|
/// the bus driver already parsed during enumeration.
|
||||||
|
pub const Endpoint = extern struct {
|
||||||
|
/// EndpointDescriptor address: direction in bit 7, number in bits 3:0.
|
||||||
|
address: u8,
|
||||||
|
/// 0 control, 1 isochronous, 2 bulk, 3 interrupt.
|
||||||
|
transfer_type: u8,
|
||||||
|
max_packet_size: u16,
|
||||||
|
interval: u8,
|
||||||
|
reserved: [3]u8 = .{ 0, 0, 0 },
|
||||||
|
};
|
||||||
|
|
||||||
|
/// open: the class driver's receive endpoint rides as the call's capability, and
|
||||||
|
/// `device_id` is the interface's assigned id (its argv[1]).
|
||||||
|
pub const OpenRequest = extern struct {
|
||||||
|
operation: u32 = @intFromEnum(Operation.open),
|
||||||
|
reserved: u32 = 0,
|
||||||
|
device_id: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The answer to open: a token scoping every later request to this device, the
|
||||||
|
/// interface's class triple (a sanity check), and its endpoints.
|
||||||
|
pub const OpenReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
endpoint_count: u32,
|
||||||
|
device_token: u64,
|
||||||
|
interface_class: u8,
|
||||||
|
interface_subclass: u8,
|
||||||
|
interface_protocol: u8,
|
||||||
|
interface_number: u8,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
endpoints: [max_reported_endpoints]Endpoint = [_]Endpoint{.{ .address = 0, .transfer_type = 0, .max_packet_size = 0, .interval = 0 }} ** max_reported_endpoints,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// control: one EP0 control transfer. `setup` is a bit-cast `usb_abi.Request`.
|
||||||
|
/// For an OUT transfer `data[0..data_length]` is sent; for an IN transfer the
|
||||||
|
/// reply carries up to `data_length` bytes back.
|
||||||
|
pub const ControlRequest = extern struct {
|
||||||
|
operation: u32 = @intFromEnum(Operation.control),
|
||||||
|
reserved: u32 = 0,
|
||||||
|
device_token: u64,
|
||||||
|
setup: [8]u8,
|
||||||
|
direction_in: u8, // 1 = device-to-host (IN), 0 = host-to-device (OUT)
|
||||||
|
reserved2: u8 = 0,
|
||||||
|
data_length: u16,
|
||||||
|
reserved3: u32 = 0,
|
||||||
|
data: [max_inline_data]u8 = [_]u8{0} ** max_inline_data,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const ControlReply = extern struct {
|
||||||
|
status: i32, // 0 success, negative on failure/stall
|
||||||
|
actual_length: u32,
|
||||||
|
data: [max_inline_data]u8 = [_]u8{0} ** max_inline_data,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// interrupt_subscribe: begin periodic IN polling of an interrupt endpoint. Each
|
||||||
|
/// report the device returns is pushed to the caller's endpoint (handed over at
|
||||||
|
/// open) as an asynchronous `InterruptReport`.
|
||||||
|
pub const InterruptSubscribeRequest = extern struct {
|
||||||
|
operation: u32 = @intFromEnum(Operation.interrupt_subscribe),
|
||||||
|
reserved: u32 = 0,
|
||||||
|
device_token: u64,
|
||||||
|
endpoint_address: u8,
|
||||||
|
reserved2: u8 = 0,
|
||||||
|
max_length: u16, // bytes to request per poll (the endpoint's max packet size)
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const InterruptSubscribeReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// bulk: one bulk IN or OUT transfer. `physical_address` is the class driver's own
|
||||||
|
/// `dma_alloc`'d buffer — the controller DMAs straight to/from it, so the bulk
|
||||||
|
/// data never crosses IPC. `endpoint_address`'s bit 7 selects IN vs OUT.
|
||||||
|
pub const BulkRequest = extern struct {
|
||||||
|
operation: u32 = @intFromEnum(Operation.bulk),
|
||||||
|
reserved: u32 = 0,
|
||||||
|
device_token: u64,
|
||||||
|
physical_address: u64,
|
||||||
|
length: u32,
|
||||||
|
endpoint_address: u8,
|
||||||
|
reserved2: u8 = 0,
|
||||||
|
reserved3: u16 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const BulkReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
actual_length: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||||
|
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||||
|
pub const InterruptReport = extern struct {
|
||||||
|
device_token: u64,
|
||||||
|
endpoint_address: u8,
|
||||||
|
length: u8,
|
||||||
|
reserved: u16 = 0,
|
||||||
|
data: [max_report_data]u8 = [_]u8{0} ** max_report_data,
|
||||||
|
};
|
||||||
|
|
||||||
|
comptime {
|
||||||
|
const std = @import("std");
|
||||||
|
// Every synchronous message must fit one IPC message; the async report must
|
||||||
|
// fit an ipc_send payload slot.
|
||||||
|
std.debug.assert(@sizeOf(ControlRequest) <= message_maximum);
|
||||||
|
std.debug.assert(@sizeOf(ControlReply) <= message_maximum);
|
||||||
|
std.debug.assert(@sizeOf(OpenReply) <= message_maximum);
|
||||||
|
std.debug.assert(@sizeOf(InterruptReport) <= 64);
|
||||||
|
}
|
||||||
@@ -17,6 +17,52 @@ const std = @import("std");
|
|||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
const protocol = runtime.device_manager_protocol;
|
const protocol = runtime.device_manager_protocol;
|
||||||
const device = runtime.device;
|
const device = runtime.device;
|
||||||
|
const usb_ids = @import("usb-ids");
|
||||||
|
const usb_abi = @import("usb-abi");
|
||||||
|
const transfer = @import("usb-transfer-protocol");
|
||||||
|
const library = @import("usb-xhci-library.zig");
|
||||||
|
|
||||||
|
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||||
|
var controller: ?library.Controller = null;
|
||||||
|
|
||||||
|
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||||
|
/// requests, signals, and the interrupt-poll timer all arrive.
|
||||||
|
var service_endpoint: runtime.ipc.Handle = 0;
|
||||||
|
|
||||||
|
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
||||||
|
/// re-armed each tick. Frequent enough for responsive input.
|
||||||
|
const poll_interval_ms: u64 = 8;
|
||||||
|
|
||||||
|
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||||
|
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||||
|
const Open = struct {
|
||||||
|
used: bool = false,
|
||||||
|
device_token: u64 = 0,
|
||||||
|
report_endpoint: usize = 0,
|
||||||
|
};
|
||||||
|
var opens = [_]Open{.{}} ** 16;
|
||||||
|
|
||||||
|
fn recordOpen(device_token: u64, report_endpoint: usize) void {
|
||||||
|
for (&opens) |*open| {
|
||||||
|
if (open.used and open.device_token == device_token) {
|
||||||
|
open.report_endpoint = report_endpoint;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (&opens) |*open| {
|
||||||
|
if (!open.used) {
|
||||||
|
open.* = .{ .used = true, .device_token = device_token, .report_endpoint = report_endpoint };
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn reportEndpointFor(device_token: u64) ?usize {
|
||||||
|
for (&opens) |*open| {
|
||||||
|
if (open.used and open.device_token == device_token) return open.report_endpoint;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
/// Format one whole log line and emit it in a single `debug_write`, so
|
/// Format one whole log line and emit it in a single `debug_write`, so
|
||||||
/// concurrent instances (one per controller) can never interleave mid-line.
|
/// concurrent instances (one per controller) can never interleave mid-line.
|
||||||
@@ -31,22 +77,22 @@ var controller_id: u64 = protocol.no_device;
|
|||||||
/// manager. Any failure returns false: the process exits cleanly, which the
|
/// manager. Any failure returns false: the process exits cleanly, which the
|
||||||
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
||||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
_ = endpoint;
|
service_endpoint = endpoint;
|
||||||
if (!device.claim(controller_id)) {
|
if (!device.claim(controller_id)) {
|
||||||
writeLine("usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
writeLine("/system/drivers/usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Fetch our own descriptor back for the controller's resources.
|
// Fetch our own descriptor back for the controller's resources.
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("usb-xhci-bus: out of memory\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: out of memory\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const total = device.enumerate(buffer);
|
const total = device.enumerate(buffer);
|
||||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
if (d.id == controller_id) break d;
|
if (d.id == controller_id) break d;
|
||||||
} else {
|
} else {
|
||||||
writeLine("usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
writeLine("/system/drivers/usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -59,19 +105,39 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
break resource;
|
break resource;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
writeLine("usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
writeLine("/system/drivers/usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
writeLine("usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
writeLine("/system/drivers/usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||||
controller_id,
|
controller_id,
|
||||||
register_window.start,
|
register_window.start,
|
||||||
register_window.len,
|
register_window.len,
|
||||||
});
|
});
|
||||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||||
_ = runtime.system.write("usb-xhci-bus: mmio_map failed\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: mmio_map failed\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Bring the controller up: reset it, stand up the command and event rings,
|
||||||
|
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||||
|
controller = library.Controller.init(register_base) orelse {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
writeLine("/system/drivers/usb-xhci-bus: controller running ({d} slots, {d}-byte contexts)\n", .{
|
||||||
|
controller.?.max_slots,
|
||||||
|
controller.?.context_size,
|
||||||
|
});
|
||||||
|
// The proof of life: a No-Op command round-trips the command ring, the event
|
||||||
|
// ring, the doorbell, and the cycle-bit bookkeeping. If this completes, the
|
||||||
|
// engine is sound; transfers build on exactly this machinery.
|
||||||
|
if (controller.?.noOpCommand()) {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: command ring running (no-op ok)\n");
|
||||||
|
} else {
|
||||||
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no-op command did not complete\n");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
// The handshake: role, protocol version, assignment — inside the manager's
|
// The handshake: role, protocol version, assignment — inside the manager's
|
||||||
// deadline (the lookup retries cover the manager still registering).
|
// deadline (the lookup retries cover the manager still registering).
|
||||||
var manager: ?runtime.ipc.Handle = null;
|
var manager: ?runtime.ipc.Handle = null;
|
||||||
@@ -81,33 +147,31 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
if (manager == null) runtime.system.sleep(20);
|
if (manager == null) runtime.system.sleep(20);
|
||||||
}
|
}
|
||||||
const h = manager orelse {
|
const h = manager orelse {
|
||||||
_ = runtime.system.write("usb-xhci-bus: no device manager to hello\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no device manager to hello\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id };
|
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id };
|
||||||
var reply: [protocol.message_maximum]u8 = undefined;
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||||
_ = runtime.system.write("usb-xhci-bus: hello call failed\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello call failed\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||||
_ = runtime.system.write("usb-xhci-bus: hello refused\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello refused\n");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
_ = runtime.system.write("usb-xhci-bus: hello acknowledged\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello acknowledged\n");
|
||||||
|
|
||||||
scanPorts(h);
|
scanPorts(h);
|
||||||
|
|
||||||
|
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
||||||
|
// re-armed on each tick in onNotification; class drivers subscribe later.
|
||||||
|
_ = runtime.system.timerOnce(service_endpoint, poll_interval_ms);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
var register_base: usize = 0;
|
var register_base: usize = 0;
|
||||||
|
|
||||||
/// One 32-bit volatile register read at `offset` from the mapped window.
|
|
||||||
fn readRegister(offset: usize) u32 {
|
|
||||||
const register: *volatile u32 = @ptrFromInt(register_base + offset);
|
|
||||||
return register.*;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||||
/// decoded to human names — the boot-log breadcrumb for what actually enumerated on
|
/// decoded to human names — the boot-log breadcrumb for what actually enumerated on
|
||||||
/// a port, the USB analog of the pci-bus class-code line. A controller may redefine
|
/// a port, the USB analog of the pci-bus class-code line. A controller may redefine
|
||||||
@@ -124,64 +188,233 @@ fn speedName(speed: u32) []const u8 {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The root-hub port scan: read the capability registers for the port count
|
/// The root-hub scan and enumeration: for each connected port, bring the device
|
||||||
/// and the operational-register offset, then one PORTSC per port. The connect
|
/// up (reset → enable slot → address), read its descriptors, and register +
|
||||||
/// bit (CCS) and the speed field reflect hardware state directly — no
|
/// report one child per interface — carrying the interface's (class, subclass,
|
||||||
/// controller reset or run needed to *see* the devices; driving them needs the
|
/// protocol) triple as identity, which is what the device manager matches a
|
||||||
/// rings (the USB track).
|
/// class driver against.
|
||||||
fn scanPorts(manager: runtime.ipc.Handle) void {
|
fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||||
// Capability registers: CAPLENGTH is byte 0 of the first dword; HCSPARAMS1
|
const engine = if (controller) |*c| c else {
|
||||||
// carries MaxPorts in bits 31:24.
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller not initialised\n");
|
||||||
const capability_length = readRegister(0) & 0xFF;
|
return;
|
||||||
const structural = readRegister(0x04);
|
};
|
||||||
const maximum_ports: u32 = structural >> 24;
|
writeLine("/system/drivers/usb-xhci-bus: {d} root-hub ports\n", .{engine.max_ports});
|
||||||
writeLine("usb-xhci-bus: {d} root-hub ports\n", .{maximum_ports});
|
|
||||||
|
|
||||||
// PORTSC registers: operational base + 0x400 + 0x10 per port (1-based).
|
|
||||||
var port: u32 = 1;
|
var port: u32 = 1;
|
||||||
var connected: u32 = 0;
|
var connected: u32 = 0;
|
||||||
while (port <= maximum_ports) : (port += 1) {
|
while (port <= engine.max_ports) : (port += 1) {
|
||||||
const port_status = readRegister(capability_length + 0x400 + 0x10 * (port - 1));
|
const port_status = engine.portStatus(port);
|
||||||
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
||||||
connected += 1;
|
connected += 1;
|
||||||
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
||||||
writeLine("usb-xhci-bus: port {d} connected — {s} (speed class {d})\n", .{ port, speedName(speed), speed });
|
writeLine("/system/drivers/usb-xhci-bus: port {d} connected — {s} (speed class {d})\n", .{ port, speedName(speed), speed });
|
||||||
|
|
||||||
const report = protocol.ChildAdded{
|
const usb_device = engine.setupDevice(port, speed) orelse {
|
||||||
.parent = controller_id,
|
writeLine("/system/drivers/usb-xhci-bus: port {d} device setup failed\n", .{port});
|
||||||
.bus_address = port,
|
|
||||||
.identity = speed,
|
|
||||||
};
|
|
||||||
var reply: [protocol.message_maximum]u8 = undefined;
|
|
||||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
|
||||||
writeLine("usb-xhci-bus: child report for port {d} failed\n", .{port});
|
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
|
if (!engine.enumerate(usb_device)) {
|
||||||
|
writeLine("/system/drivers/usb-xhci-bus: port {d} enumeration failed\n", .{port});
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
writeLine("/system/drivers/usb-xhci-bus: port {d} device vendor 0x{x:0>4} product 0x{x:0>4}, {d} interface(s)\n", .{
|
||||||
|
port,
|
||||||
|
usb_device.device_descriptor.vendor_id,
|
||||||
|
usb_device.device_descriptor.product_id,
|
||||||
|
usb_device.interface_count,
|
||||||
|
});
|
||||||
|
|
||||||
|
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||||
|
// Record the id each interface was registered as, so a class driver
|
||||||
|
// opening the interface (by that id) resolves to it.
|
||||||
|
if (reportInterface(manager, port, interface.*)) |registered| {
|
||||||
|
interface.registered_device_id = registered;
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (connected == 0) _ = runtime.system.write("usb-xhci-bus: no devices connected\n");
|
if (connected == 0) _ = runtime.system.write("/system/drivers/usb-xhci-bus: no devices connected\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// No bus protocol to serve yet — transfer requests arrive with the USB track.
|
/// Register one interface as a resource-less child of the controller and report
|
||||||
|
/// it to the device manager. The identity is the packed USB class triple, so the
|
||||||
|
/// manager can match a class driver (HID keyboard, mouse, mass storage); the
|
||||||
|
/// registered device id becomes that driver's argv[1] assignment. Returns the
|
||||||
|
/// registered device id, or null if registration or the report failed.
|
||||||
|
fn reportInterface(manager: runtime.ipc.Handle, port: u32, interface: library.InterfaceInfo) ?u64 {
|
||||||
|
const identity = usb_ids.packTriple(interface.class, interface.subclass, interface.protocol);
|
||||||
|
|
||||||
|
// A USB device is reached through its controller, not by MMIO, so the child
|
||||||
|
// carries no resources; register() allows that. Its bus-local identity — the
|
||||||
|
// (port, interface) address, written as a short "P<port>I<interface>" tag in
|
||||||
|
// the hid field — makes each interface a distinct kernel node (the register
|
||||||
|
// dedup keys on class/pci_class/hid/resources, all otherwise identical here)
|
||||||
|
// and keeps re-registration idempotent across a bus restart: the same port
|
||||||
|
// and interface always map back to the same device id.
|
||||||
|
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||||
|
descriptor.class = @intFromEnum(device.DeviceClass.usb_device);
|
||||||
|
descriptor.pci_class = device.no_pci_class;
|
||||||
|
descriptor.resource_count = 0;
|
||||||
|
var hid_buffer: [8]u8 = undefined;
|
||||||
|
const hid_text = std.fmt.bufPrint(&hid_buffer, "P{d}I{d}", .{ port, interface.number }) catch "";
|
||||||
|
descriptor.hid_len = hid_text.len;
|
||||||
|
@memcpy(descriptor.hid[0..hid_text.len], hid_text);
|
||||||
|
const registered = device.register(controller_id, &descriptor) orelse {
|
||||||
|
writeLine("/system/drivers/usb-xhci-bus: register refused for port {d} interface {d}\n", .{ port, interface.number });
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
|
||||||
|
const report = protocol.ChildAdded{
|
||||||
|
.parent = controller_id,
|
||||||
|
.bus_address = (@as(u64, port) << 8) | interface.number,
|
||||||
|
.identity = identity,
|
||||||
|
.device_id = registered,
|
||||||
|
};
|
||||||
|
var reply: [protocol.message_maximum]u8 = undefined;
|
||||||
|
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||||
|
writeLine("/system/drivers/usb-xhci-bus: child report for port {d} interface {d} failed\n", .{ port, interface.number });
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
writeLine("/system/drivers/usb-xhci-bus: port {d} interface {d} class {d}/{d}/{d} registered as device {d}\n", .{
|
||||||
|
port,
|
||||||
|
interface.number,
|
||||||
|
interface.class,
|
||||||
|
interface.subclass,
|
||||||
|
interface.protocol,
|
||||||
|
registered,
|
||||||
|
});
|
||||||
|
return registered;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serve the USB transfer protocol: a class driver opens its device, then issues
|
||||||
|
/// control / interrupt-subscribe / bulk requests against it.
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
_ = message;
|
|
||||||
_ = reply;
|
|
||||||
_ = sender;
|
_ = sender;
|
||||||
_ = capability;
|
if (message.len < 4) return 0;
|
||||||
return 0;
|
const operation = std.mem.readInt(u32, message[0..4], .little);
|
||||||
|
return switch (operation) {
|
||||||
|
@intFromEnum(transfer.Operation.open) => handleOpen(message, reply, capability),
|
||||||
|
@intFromEnum(transfer.Operation.control) => handleControl(message, reply),
|
||||||
|
@intFromEnum(transfer.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||||
|
@intFromEnum(transfer.Operation.bulk) => handleBulk(message, reply),
|
||||||
|
else => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeReply(reply: []u8, value: anytype) usize {
|
||||||
|
const bytes = std.mem.asBytes(&value);
|
||||||
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
|
return bytes.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// open: resolve the assigned device id to an interface, remember the caller's
|
||||||
|
/// endpoint (for interrupt reports), and answer with a device token + the
|
||||||
|
/// interface's endpoints so the class driver need not re-read the config.
|
||||||
|
fn handleOpen(message: []const u8, reply: []u8, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
if (message.len < @sizeOf(transfer.OpenRequest)) return writeReply(reply, transfer.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||||
|
const request = std.mem.bytesToValue(transfer.OpenRequest, message[0..@sizeOf(transfer.OpenRequest)]);
|
||||||
|
const engine = if (controller) |*c| c else return writeReply(reply, transfer.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||||
|
const found = engine.findInterface(request.device_id) orelse return writeReply(reply, transfer.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||||
|
|
||||||
|
if (capability) |endpoint| recordOpen(request.device_id, endpoint);
|
||||||
|
|
||||||
|
var open_reply = transfer.OpenReply{
|
||||||
|
.status = 0,
|
||||||
|
.endpoint_count = found.interface.endpoint_count,
|
||||||
|
.device_token = request.device_id,
|
||||||
|
.interface_class = found.interface.class,
|
||||||
|
.interface_subclass = found.interface.subclass,
|
||||||
|
.interface_protocol = found.interface.protocol,
|
||||||
|
.interface_number = found.interface.number,
|
||||||
|
};
|
||||||
|
const count = @min(found.interface.endpoint_count, transfer.max_reported_endpoints);
|
||||||
|
for (found.interface.endpoints[0..count], 0..) |endpoint, index| {
|
||||||
|
open_reply.endpoints[index] = .{
|
||||||
|
.address = endpoint.address,
|
||||||
|
.transfer_type = endpoint.transfer_type,
|
||||||
|
.max_packet_size = endpoint.max_packet_size,
|
||||||
|
.interval = endpoint.interval,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
return writeReply(reply, open_reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// control: one EP0 control transfer, small data inline both ways.
|
||||||
|
fn handleControl(message: []const u8, reply: []u8) usize {
|
||||||
|
if (message.len < @sizeOf(transfer.ControlRequest)) return writeReply(reply, transfer.ControlReply{ .status = -1, .actual_length = 0 });
|
||||||
|
const request = std.mem.bytesToValue(transfer.ControlRequest, message[0..@sizeOf(transfer.ControlRequest)]);
|
||||||
|
const engine = if (controller) |*c| c else return writeReply(reply, transfer.ControlReply{ .status = -1, .actual_length = 0 });
|
||||||
|
const found = engine.findInterface(request.device_token) orelse return writeReply(reply, transfer.ControlReply{ .status = -1, .actual_length = 0 });
|
||||||
|
|
||||||
|
const setup = std.mem.bytesToValue(usb_abi.Request, &request.setup);
|
||||||
|
const direction_in = request.direction_in != 0;
|
||||||
|
const data_length = @min(request.data_length, transfer.max_inline_data);
|
||||||
|
var data: [transfer.max_inline_data]u8 = undefined;
|
||||||
|
if (!direction_in) @memcpy(data[0..data_length], request.data[0..data_length]);
|
||||||
|
|
||||||
|
const ok = engine.controlTransfer(found.device, setup, data[0..data_length], direction_in);
|
||||||
|
var control_reply = transfer.ControlReply{ .status = if (ok) 0 else -1, .actual_length = if (ok) data_length else 0 };
|
||||||
|
if (ok and direction_in) @memcpy(control_reply.data[0..data_length], data[0..data_length]);
|
||||||
|
return writeReply(reply, control_reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// interrupt_subscribe: arm periodic IN polling; reports flow back asynchronously.
|
||||||
|
fn handleSubscribe(message: []const u8, reply: []u8) usize {
|
||||||
|
if (message.len < @sizeOf(transfer.InterruptSubscribeRequest)) return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||||
|
const request = std.mem.bytesToValue(transfer.InterruptSubscribeRequest, message[0..@sizeOf(transfer.InterruptSubscribeRequest)]);
|
||||||
|
const engine = if (controller) |*c| c else return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||||
|
const found = engine.findInterface(request.device_token) orelse return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||||
|
const endpoint = library.Controller.endpointForAddress(found.interface, request.endpoint_address) orelse return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||||
|
const report_endpoint = reportEndpointFor(request.device_token) orelse return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||||
|
const ok = engine.subscribeInterrupt(found.device, endpoint, request.device_token, report_endpoint);
|
||||||
|
return writeReply(reply, transfer.InterruptSubscribeReply{ .status = if (ok) 0 else -1 });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// bulk: one bulk transfer to/from the class driver's own DMA buffer (by physical
|
||||||
|
/// address), so sector-sized data never crosses IPC.
|
||||||
|
fn handleBulk(message: []const u8, reply: []u8) usize {
|
||||||
|
if (message.len < @sizeOf(transfer.BulkRequest)) return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||||
|
const request = std.mem.bytesToValue(transfer.BulkRequest, message[0..@sizeOf(transfer.BulkRequest)]);
|
||||||
|
const engine = if (controller) |*c| c else return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||||
|
const found = engine.findInterface(request.device_token) orelse return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||||
|
const endpoint = library.Controller.endpointForAddress(found.interface, request.endpoint_address) orelse return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||||
|
const transferred = engine.bulkTransfer(found.device, endpoint, request.physical_address, request.length);
|
||||||
|
return writeReply(reply, transfer.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
||||||
|
/// each to the class driver that subscribed, then re-arm the timer.
|
||||||
|
fn onNotification(badge: u64) void {
|
||||||
|
if (badge & runtime.ipc.notify_timer_bit == 0) return;
|
||||||
|
if (controller) |*engine| {
|
||||||
|
engine.pump();
|
||||||
|
while (engine.takeReport()) |report| {
|
||||||
|
var message = transfer.InterruptReport{
|
||||||
|
.device_token = report.device_token,
|
||||||
|
.endpoint_address = report.endpoint_address,
|
||||||
|
.length = @intCast(@min(report.length, transfer.max_report_data)),
|
||||||
|
};
|
||||||
|
const n = @min(report.length, transfer.max_report_data);
|
||||||
|
@memcpy(message.data[0..n], report.data[0..n]);
|
||||||
|
_ = runtime.ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = runtime.system.timerOnce(service_endpoint, poll_interval_ms);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
const argument = init.arguments.get(1) orelse {
|
const argument = init.arguments.get(1) orelse {
|
||||||
_ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n");
|
_ = runtime.system.write("/system/drivers/usb-xhci-bus: missing controller device id (argv[1])\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
writeLine("/system/drivers/usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
runtime.service.run(protocol.message_maximum, .{
|
runtime.service.run(transfer.message_maximum, .{
|
||||||
|
.service = .usb_bus,
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
|
.on_notification = onNotification,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,139 @@
|
|||||||
|
//! The virtio-gpu control protocol — the command/response structs the driver exchanges with
|
||||||
|
//! the device over its control virtqueue (virtio spec, "GPU Device"). `extern` structs, so
|
||||||
|
//! the layout matches the little-endian wire format exactly. Host-tested for size. See
|
||||||
|
//! docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Control command / response types (virtio_gpu_ctrl_type). Commands are 0x01xx, responses
|
||||||
|
/// 0x11xx (ok) / 0x12xx (error).
|
||||||
|
pub const CmdType = enum(u32) {
|
||||||
|
get_display_info = 0x0100,
|
||||||
|
resource_create_2d = 0x0101,
|
||||||
|
resource_unref = 0x0102,
|
||||||
|
set_scanout = 0x0103,
|
||||||
|
resource_flush = 0x0104,
|
||||||
|
transfer_to_host_2d = 0x0105,
|
||||||
|
resource_attach_backing = 0x0106,
|
||||||
|
resource_detach_backing = 0x0107,
|
||||||
|
get_edid = 0x010a,
|
||||||
|
|
||||||
|
resp_ok_nodata = 0x1100,
|
||||||
|
resp_ok_display_info = 0x1101,
|
||||||
|
resp_ok_edid = 0x1104,
|
||||||
|
resp_err_unspec = 0x1200,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Set in a command's `flags` to request a fence; the device echoes `fence_id` in the
|
||||||
|
/// response and does not report completion until the command's effects are visible.
|
||||||
|
pub const flag_fence: u32 = 1 << 0;
|
||||||
|
|
||||||
|
/// VIRTIO_GPU_F_EDID — device feature bit 1 (the low feature word): the device answers the
|
||||||
|
/// `get_edid` command. Negotiate it only when the device offers it.
|
||||||
|
pub const feature_edid: u32 = 1 << 1;
|
||||||
|
|
||||||
|
/// virtio_gpu_ctrl_hdr — the header on every command and response.
|
||||||
|
pub const CtrlHdr = extern struct {
|
||||||
|
type: u32,
|
||||||
|
flags: u32 = 0,
|
||||||
|
fence_id: u64 = 0,
|
||||||
|
ctx_id: u32 = 0,
|
||||||
|
ring_idx: u8 = 0,
|
||||||
|
padding: [3]u8 = .{ 0, 0, 0 },
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Rect = extern struct {
|
||||||
|
x: u32,
|
||||||
|
y: u32,
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 2D pixel formats. QEMU's virtio-gpu host default is B8G8R8X8 (matches our bgrx).
|
||||||
|
pub const format_b8g8r8x8_unorm: u32 = 2;
|
||||||
|
pub const format_r8g8b8x8_unorm: u32 = 134;
|
||||||
|
|
||||||
|
pub const ResourceCreate2d = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
resource_id: u32,
|
||||||
|
format: u32,
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One scatter-gather entry of a resource's guest backing (a physical span).
|
||||||
|
pub const MemEntry = extern struct {
|
||||||
|
addr: u64,
|
||||||
|
length: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Header for RESOURCE_ATTACH_BACKING; `nr_entries` `MemEntry` follow it inline.
|
||||||
|
pub const ResourceAttachBacking = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
resource_id: u32,
|
||||||
|
nr_entries: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const SetScanout = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
rect: Rect,
|
||||||
|
scanout_id: u32,
|
||||||
|
resource_id: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const ResourceFlush = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
rect: Rect,
|
||||||
|
resource_id: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Copy the guest backing into the host resource for `rect` (2D resources must transfer
|
||||||
|
/// before a flush shows the update).
|
||||||
|
pub const TransferToHost2d = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
rect: Rect,
|
||||||
|
offset: u64,
|
||||||
|
resource_id: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const max_scanouts = 16;
|
||||||
|
|
||||||
|
pub const DisplayOne = extern struct {
|
||||||
|
rect: Rect,
|
||||||
|
enabled: u32,
|
||||||
|
flags: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const RespDisplayInfo = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
pmodes: [max_scanouts]DisplayOne,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const GetEdid = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
scanout: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const RespEdid = extern struct {
|
||||||
|
hdr: CtrlHdr,
|
||||||
|
size: u32,
|
||||||
|
padding: u32 = 0,
|
||||||
|
edid: [1024]u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
test "virtio-gpu struct sizes match the wire layout" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 24), @sizeOf(CtrlHdr));
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Rect));
|
||||||
|
try std.testing.expectEqual(@as(usize, 40), @sizeOf(ResourceCreate2d));
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(MemEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(ResourceAttachBacking));
|
||||||
|
try std.testing.expectEqual(@as(usize, 48), @sizeOf(SetScanout));
|
||||||
|
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ResourceFlush));
|
||||||
|
try std.testing.expectEqual(@as(usize, 56), @sizeOf(TransferToHost2d));
|
||||||
|
try std.testing.expectEqual(@as(usize, 24 + 4 + 4 + 1024), @sizeOf(RespEdid));
|
||||||
|
}
|
||||||
@@ -0,0 +1,622 @@
|
|||||||
|
//! /system/drivers/virtio-gpu — the virtio-gpu (virtio 1.0, modern PCI) display driver.
|
||||||
|
//! The device manager spawns it for the display/other PCI function (class 0x0380) whose
|
||||||
|
//! config space says vendor 0x1AF4 / device 0x1050; this instance claims that device and
|
||||||
|
//! brings up a single 2D scanout.
|
||||||
|
//!
|
||||||
|
//! V3 (this increment): the whole path end to end, proven from serial without a screenshot.
|
||||||
|
//! Claim the function, map its config space (resource 0) and the BAR that carries the
|
||||||
|
//! virtio structures, walk the vendor capabilities to find common-config / notify, reset
|
||||||
|
//! and negotiate VERSION_1, stand up the control virtqueue in DMA memory, then drive the
|
||||||
|
//! GPU: RESOURCE_CREATE_2D → ATTACH_BACKING (a coherent DMA buffer) → SET_SCANOUT, paint a
|
||||||
|
//! known test pattern, TRANSFER_TO_HOST_2D → RESOURCE_FLUSH, and **wait for the device's
|
||||||
|
//! used-ring ack**. Reading the backing back confirms it is CPU-visible; the ack confirms
|
||||||
|
//! the device consumed the frame. The compositor backend, hot-attach, mode-set/EDID, and
|
||||||
|
//! restart/re-attach are V4–V6. See docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const mmio = @import("mmio");
|
||||||
|
const device = runtime.device;
|
||||||
|
const dma = runtime.dma;
|
||||||
|
const shm = runtime.shm;
|
||||||
|
const system = runtime.system;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const dp = runtime.display_protocol;
|
||||||
|
const sp = runtime.scanout_protocol;
|
||||||
|
const dm = runtime.device_manager_protocol;
|
||||||
|
const vp = @import("virtio-pci.zig");
|
||||||
|
const vg = @import("virtio-gpu-protocol.zig");
|
||||||
|
|
||||||
|
/// The DisplayFormat (device-abi) our B8G8R8X8 scanout resource presents: bgrx = 1. Handed to
|
||||||
|
/// the compositor in the announce so it packs colours in the surface's byte order.
|
||||||
|
const display_format_bgrx: u32 = 1;
|
||||||
|
|
||||||
|
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
|
||||||
|
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
|
||||||
|
const virtio_vendor: u16 = 0x1AF4;
|
||||||
|
const virtio_gpu_device: u16 = 0x1050;
|
||||||
|
|
||||||
|
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
||||||
|
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
||||||
|
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
||||||
|
/// the compositor is told in the announce. Kept modest so the backing is an easy contiguous run.
|
||||||
|
const max_width: u32 = 800;
|
||||||
|
const max_height: u32 = 600;
|
||||||
|
const scanout_bytes: usize = @as(usize, max_width) * max_height * 4;
|
||||||
|
const resource_id: u32 = 1;
|
||||||
|
|
||||||
|
/// The modes this scanout offers (all ≤ max). The first is the mode it comes up in.
|
||||||
|
const Mode = struct { width: u32, height: u32 };
|
||||||
|
const offered_modes = [_]Mode{ .{ .width = 640, .height = 480 }, .{ .width = 800, .height = 600 } };
|
||||||
|
|
||||||
|
/// The active mode — the scanout rectangle within the max-sized surface. Changed by `set_mode`.
|
||||||
|
var current_width: u32 = offered_modes[0].width;
|
||||||
|
var current_height: u32 = offered_modes[0].height;
|
||||||
|
|
||||||
|
/// Monotonic fence id for fenced (vsync) flushes; the device signals the fence when the flush
|
||||||
|
/// is complete, which its used-ring ack already gates our synchronous present on.
|
||||||
|
var fence_next: u64 = 1;
|
||||||
|
|
||||||
|
/// Whether the device offered VIRTIO_GPU_F_EDID, so `get_edid` is worth issuing.
|
||||||
|
var edid_available = false;
|
||||||
|
|
||||||
|
/// The control virtqueue. We drive it synchronously — one command, notify, poll the used
|
||||||
|
/// ring — so a depth of 16 is ample; we ask the device to shrink to it (virtio 1.0 lets the
|
||||||
|
/// driver reduce queue_size), keeping the whole ring inside one page.
|
||||||
|
const queue_size: u16 = 16;
|
||||||
|
const desc_offset: usize = 0; // 16 * 16 = 256 bytes
|
||||||
|
const avail_offset: usize = 256; // flags + idx + ring[16] + used_event = 38 bytes
|
||||||
|
const used_offset: usize = 1024; // flags + idx + ring[16] + avail_event = 134 bytes
|
||||||
|
|
||||||
|
/// The command scratch: the request the device reads, then its response, in one DMA page.
|
||||||
|
const request_offset: usize = 0;
|
||||||
|
const response_offset: usize = 2048;
|
||||||
|
|
||||||
|
var device_id: u64 = 0;
|
||||||
|
|
||||||
|
// Mapped virtio structures (virtual addresses into the device's BAR).
|
||||||
|
var common_base: usize = 0;
|
||||||
|
var notify_base: usize = 0;
|
||||||
|
var notify_multiplier: u32 = 0;
|
||||||
|
var notify_addr: usize = 0;
|
||||||
|
|
||||||
|
// Per-BAR mapping cache: several capabilities usually share one BAR, and mmio_map must not
|
||||||
|
// be asked to map the same resource twice.
|
||||||
|
var bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 };
|
||||||
|
|
||||||
|
// DMA memory: the virtqueue rings and the command scratch.
|
||||||
|
var ring: dma.Region = undefined;
|
||||||
|
var command: dma.Region = undefined;
|
||||||
|
|
||||||
|
// The scanout backing is a **shared** (shm) region, not DMA: cacheable so the compositor
|
||||||
|
// composites into it cheaply (x86 DMA is coherent, so the device still sees the writes), and
|
||||||
|
// shareable so the same physical pages the device scans out of are the ones the compositor
|
||||||
|
// paints. The driver keeps the capability to hand to the compositor in the announce.
|
||||||
|
var surface: shm.Region = undefined;
|
||||||
|
|
||||||
|
// Split-virtqueue producer/consumer shadows.
|
||||||
|
var avail_shadow: u16 = 0;
|
||||||
|
var used_shadow: u16 = 0;
|
||||||
|
|
||||||
|
/// Format one whole log line and emit it in a single `write`, so this driver's output can
|
||||||
|
/// never interleave mid-line with the other drivers the manager runs concurrently.
|
||||||
|
fn log(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [160]u8 = undefined;
|
||||||
|
_ = system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- common-config register access (little-endian MMIO at `common_base`) ---------------
|
||||||
|
|
||||||
|
fn cfgRead(comptime T: type, comptime field: []const u8) T {
|
||||||
|
return mmio.read(T, common_base + @offsetOf(vp.CommonCfg, field));
|
||||||
|
}
|
||||||
|
fn cfgWrite(comptime T: type, comptime field: []const u8, value: T) void {
|
||||||
|
mmio.write(T, common_base + @offsetOf(vp.CommonCfg, field), value);
|
||||||
|
}
|
||||||
|
/// Write a 64-bit common-config register as two 32-bit halves (low then high) — the widest
|
||||||
|
/// access every virtio-pci host is required to accept for the queue-address registers.
|
||||||
|
fn cfgWrite64(comptime field: []const u8, value: u64) void {
|
||||||
|
const at = common_base + @offsetOf(vp.CommonCfg, field);
|
||||||
|
mmio.write(u32, at, @truncate(value));
|
||||||
|
mmio.write(u32, at + 4, @truncate(value >> 32));
|
||||||
|
}
|
||||||
|
fn orStatus(bit: u8) void {
|
||||||
|
cfgWrite(u8, "device_status", cfgRead(u8, "device_status") | bit);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- PCI config-space capability walk (config space is resource 0) ---------------------
|
||||||
|
|
||||||
|
/// Map the BAR numbered `bar` (0..5) and return its virtual base, correlating the BAR's
|
||||||
|
/// physical address (read from config space) with one of our device resources — because a
|
||||||
|
/// virtio capability names a BAR *number*, while `mmio_map` takes a *resource index* (and
|
||||||
|
/// resource 0 is config space, so BAR resources are re-numbered and gaps skipped).
|
||||||
|
fn mapBar(config: usize, descriptor: *const device.DeviceDescriptor, bar: u8) ?usize {
|
||||||
|
if (bar >= 6) return null;
|
||||||
|
if (bar_virtual[bar] != 0) return bar_virtual[bar];
|
||||||
|
|
||||||
|
const low = mmio.read(u32, config + 0x10 + @as(usize, bar) * 4);
|
||||||
|
if (low & 0x1 != 0) return null; // an I/O-space BAR — virtio structures are in memory BARs
|
||||||
|
var base: u64 = low & 0xFFFF_FFF0;
|
||||||
|
if ((low & 0x6) == 0x4) { // 64-bit memory BAR: the high half is the next dword
|
||||||
|
const high = mmio.read(u32, config + 0x10 + (@as(usize, bar) + 1) * 4);
|
||||||
|
base |= @as(u64, high) << 32;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||||
|
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||||
|
const v = device.mmioMap(device_id, index) orelse return null;
|
||||||
|
bar_virtual[bar] = v;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
log("virtio-gpu: BAR {d} (physical 0x{x}) is not a mapped resource\n", .{ bar, base });
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Walk the PCI capability list from mapped config space, recording the common-config and
|
||||||
|
/// notify structures (the only two V3 needs). Returns false if either is missing.
|
||||||
|
fn walkCapabilities(config: usize, descriptor: *const device.DeviceDescriptor) bool {
|
||||||
|
if (mmio.read(u16, config + 0x06) & 0x10 == 0) { // Status bit 4: capabilities list present
|
||||||
|
log("virtio-gpu: device has no PCI capability list\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
var cap: u8 = @as(u8, @truncate(mmio.read(u8, config + 0x34))) & 0xFC;
|
||||||
|
var guard: u32 = 0;
|
||||||
|
while (cap != 0 and guard < 48) : (guard += 1) {
|
||||||
|
const at = config + cap;
|
||||||
|
const id = mmio.read(u8, at + 0);
|
||||||
|
const next = mmio.read(u8, at + 1) & 0xFC;
|
||||||
|
// Only map BARs for the structures V3 uses (common + notify). The other virtio
|
||||||
|
// capabilities (isr, device, and especially the cfg_pci back-door, which carries a
|
||||||
|
// placeholder bar=0/offset=0) reference BARs we never touch, so mapping them would
|
||||||
|
// just log spurious "not a mapped resource" noise.
|
||||||
|
if (id == vp.pci_cap_vendor) {
|
||||||
|
const cfg_type = mmio.read(u8, at + 3);
|
||||||
|
if (cfg_type == vp.cfg_common or cfg_type == vp.cfg_notify) {
|
||||||
|
const bar = mmio.read(u8, at + 4);
|
||||||
|
const offset = mmio.read(u32, at + 8);
|
||||||
|
if (mapBar(config, descriptor, bar)) |bar_base| {
|
||||||
|
if (cfg_type == vp.cfg_common) {
|
||||||
|
common_base = bar_base + offset;
|
||||||
|
} else {
|
||||||
|
notify_base = bar_base + offset;
|
||||||
|
notify_multiplier = mmio.read(u32, at + 16); // virtio_pci_notify_cap tail
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
cap = next;
|
||||||
|
}
|
||||||
|
if (common_base == 0 or notify_base == 0) {
|
||||||
|
log("virtio-gpu: missing common-config or notify capability\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the control virtqueue -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Publish the two-descriptor chain (request read by the device, response written by it),
|
||||||
|
/// notify the control queue, and wait for the device to return the buffer on the used ring.
|
||||||
|
fn submit(request_len: usize, response_len: usize) bool {
|
||||||
|
const desc: [*]vp.Desc = @ptrFromInt(ring.virtual + desc_offset);
|
||||||
|
desc[0] = .{
|
||||||
|
.addr = command.physical + request_offset,
|
||||||
|
.len = @intCast(request_len),
|
||||||
|
.flags = vp.desc_flag_next,
|
||||||
|
.next = 1,
|
||||||
|
};
|
||||||
|
desc[1] = .{
|
||||||
|
.addr = command.physical + response_offset,
|
||||||
|
.len = @intCast(response_len),
|
||||||
|
.flags = vp.desc_flag_write,
|
||||||
|
.next = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
const avail_ring: [*]u16 = @ptrFromInt(ring.virtual + avail_offset + 4);
|
||||||
|
avail_ring[avail_shadow % queue_size] = 0; // head of the chain is descriptor 0
|
||||||
|
mmio.wmb();
|
||||||
|
avail_shadow +%= 1;
|
||||||
|
mmio.write(u16, ring.virtual + avail_offset + 2, avail_shadow); // avail.idx
|
||||||
|
mmio.wmb();
|
||||||
|
|
||||||
|
mmio.write(u16, notify_addr, 0); // ring the control queue's doorbell
|
||||||
|
return waitUsed();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spin, then sleep-poll, on the used-ring index until the device advances it. QEMU
|
||||||
|
/// processes the notify on its own thread, so the ack usually lands immediately; the sleep
|
||||||
|
/// fallback covers a device that defers it without burning the CPU.
|
||||||
|
fn waitUsed() bool {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (tries < 2000) : (tries += 1) {
|
||||||
|
mmio.rmb();
|
||||||
|
const idx = mmio.read(u16, ring.virtual + used_offset + 2); // used.idx
|
||||||
|
if (idx != used_shadow) {
|
||||||
|
used_shadow = idx;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (tries > 8) system.sleep(1);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The type field of the response the device wrote — `resp_ok_nodata` on success.
|
||||||
|
fn responseType() u32 {
|
||||||
|
const response: *vg.CtrlHdr = @ptrFromInt(command.virtual + response_offset);
|
||||||
|
return response.type;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Submit a command whose response is a bare header, returning its response type (0 if the
|
||||||
|
/// device never acked).
|
||||||
|
fn command_nodata(request_len: usize) u32 {
|
||||||
|
if (!submit(request_len, @sizeOf(vg.CtrlHdr))) return 0;
|
||||||
|
return responseType();
|
||||||
|
}
|
||||||
|
|
||||||
|
const ok_nodata: u32 = @intFromEnum(vg.CmdType.resp_ok_nodata);
|
||||||
|
|
||||||
|
fn requestAt(comptime T: type) *T {
|
||||||
|
return @ptrFromInt(command.virtual + request_offset);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A deterministic, recognisable pixel so a read-back is a real check, not a tautology.
|
||||||
|
fn testPixel(index: u32) u32 {
|
||||||
|
return 0xFF00_0000 | (index *% 0x9E37_79B1);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- bring-up --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
|
_ = endpoint;
|
||||||
|
if (!device.claim(device_id)) {
|
||||||
|
log("virtio-gpu: unable to claim device {d}\n", .{device_id});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||||
|
const total = device.enumerate(&descriptors);
|
||||||
|
const descriptor = for (descriptors[0..@min(total, descriptors.len)]) |*d| {
|
||||||
|
if (d.id == device_id) break d;
|
||||||
|
} else {
|
||||||
|
log("virtio-gpu: device {d} not in the device tree\n", .{device_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Config space is resource 0. Confirm it really is a virtio-gpu, then enable memory-space
|
||||||
|
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
||||||
|
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
||||||
|
const config = device.mmioMap(device_id, 0) orelse {
|
||||||
|
log("virtio-gpu: config-space map failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const vendor = mmio.read(u16, config + 0x00);
|
||||||
|
const dev = mmio.read(u16, config + 0x02);
|
||||||
|
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
||||||
|
log("virtio-gpu: not a virtio-gpu (vendor 0x{x} device 0x{x})\n", .{ vendor, dev });
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
mmio.write(u16, config + 0x04, mmio.read(u16, config + 0x04) | 0x06); // MEM + bus master
|
||||||
|
|
||||||
|
if (!walkCapabilities(config, descriptor)) return false;
|
||||||
|
|
||||||
|
// Reset, then the modern feature handshake: acknowledge, take driver ownership, require
|
||||||
|
// VERSION_1 and offer nothing else, and confirm the device accepts that.
|
||||||
|
cfgWrite(u8, "device_status", 0);
|
||||||
|
orStatus(vp.status_acknowledge);
|
||||||
|
orStatus(vp.status_driver);
|
||||||
|
|
||||||
|
// Low feature word (device-specific): note whether the device offers EDID (bit 1).
|
||||||
|
cfgWrite(u32, "device_feature_select", 0);
|
||||||
|
edid_available = cfgRead(u32, "device_feature") & vg.feature_edid != 0;
|
||||||
|
// High feature word: VERSION_1 (bit 32) is required for a modern device.
|
||||||
|
cfgWrite(u32, "device_feature_select", vp.feature_version_1_word);
|
||||||
|
if (cfgRead(u32, "device_feature") & vp.feature_version_1_bit == 0) {
|
||||||
|
log("virtio-gpu: device does not offer VERSION_1 (not a modern device)\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// Accept exactly VERSION_1, plus EDID when the device offered it (never a feature it didn't).
|
||||||
|
cfgWrite(u32, "driver_feature_select", 0);
|
||||||
|
cfgWrite(u32, "driver_feature", if (edid_available) vg.feature_edid else 0);
|
||||||
|
cfgWrite(u32, "driver_feature_select", vp.feature_version_1_word);
|
||||||
|
cfgWrite(u32, "driver_feature", vp.feature_version_1_bit);
|
||||||
|
orStatus(vp.status_features_ok);
|
||||||
|
if (cfgRead(u8, "device_status") & vp.status_features_ok == 0) {
|
||||||
|
log("virtio-gpu: device rejected the negotiated features\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Stand up the control virtqueue (queue 0) in coherent DMA memory.
|
||||||
|
cfgWrite(u16, "queue_select", 0);
|
||||||
|
const device_qsize = cfgRead(u16, "queue_size");
|
||||||
|
if (device_qsize < queue_size) {
|
||||||
|
log("virtio-gpu: control queue too small ({d})\n", .{device_qsize});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
ring = dma.alloc(4096, dma.coherent) orelse {
|
||||||
|
log("virtio-gpu: virtqueue allocation failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
command = dma.alloc(4096, dma.coherent) orelse {
|
||||||
|
log("virtio-gpu: command-buffer allocation failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
mmio.write(u16, ring.virtual + avail_offset, 1); // VIRTQ_AVAIL_F_NO_INTERRUPT: we poll
|
||||||
|
cfgWrite(u16, "queue_size", queue_size);
|
||||||
|
cfgWrite64("queue_desc", ring.physical + desc_offset);
|
||||||
|
cfgWrite64("queue_driver", ring.physical + avail_offset);
|
||||||
|
cfgWrite64("queue_device", ring.physical + used_offset);
|
||||||
|
cfgWrite(u16, "queue_msix_vector", 0xFFFF); // VIRTIO_MSI_NO_VECTOR
|
||||||
|
cfgWrite(u16, "queue_enable", 1);
|
||||||
|
|
||||||
|
cfgWrite(u16, "queue_select", 0);
|
||||||
|
notify_addr = notify_base + @as(usize, cfgRead(u16, "queue_notify_off")) * notify_multiplier;
|
||||||
|
|
||||||
|
orStatus(vp.status_driver_ok);
|
||||||
|
|
||||||
|
// Drive the GPU: create a 2D resource at the *max* mode, back it with a shared surface, and
|
||||||
|
// scan out the current-mode rectangle within it.
|
||||||
|
{
|
||||||
|
const request = requestAt(vg.ResourceCreate2d);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_create_2d) },
|
||||||
|
.resource_id = resource_id,
|
||||||
|
.format = vg.format_b8g8r8x8_unorm,
|
||||||
|
.width = max_width,
|
||||||
|
.height = max_height,
|
||||||
|
};
|
||||||
|
if (command_nodata(@sizeOf(vg.ResourceCreate2d)) != ok_nodata) {
|
||||||
|
log("virtio-gpu: resource_create_2d failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Back the resource with a shared (shm) surface, so the compositor and the device work
|
||||||
|
// the same physical pages. The device needs the guest-physical base for attach_backing.
|
||||||
|
surface = shm.create(scanout_bytes) orelse {
|
||||||
|
log("virtio-gpu: scanout surface allocation failed\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
const surface_physical = shm.physical(surface.handle) orelse {
|
||||||
|
log("virtio-gpu: could not resolve the scanout surface physical address\n", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
{
|
||||||
|
const request = requestAt(vg.ResourceAttachBacking);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_attach_backing) },
|
||||||
|
.resource_id = resource_id,
|
||||||
|
.nr_entries = 1,
|
||||||
|
};
|
||||||
|
const entry: *vg.MemEntry = @ptrFromInt(command.virtual + request_offset + @sizeOf(vg.ResourceAttachBacking));
|
||||||
|
entry.* = .{ .addr = surface_physical, .length = @intCast(scanout_bytes) };
|
||||||
|
if (command_nodata(@sizeOf(vg.ResourceAttachBacking) + @sizeOf(vg.MemEntry)) != ok_nodata) {
|
||||||
|
log("virtio-gpu: resource_attach_backing failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!setScanoutRect()) {
|
||||||
|
log("virtio-gpu: set_scanout failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
log("virtio-gpu: scanout {d}x{d} online\n", .{ current_width, current_height });
|
||||||
|
|
||||||
|
// Hello the device manager so it counts us as up (and does not stop us at the hello
|
||||||
|
// deadline). A restarted instance re-hellos here and re-announces below — the compositor
|
||||||
|
// re-attaches to the fresh scanout (V6).
|
||||||
|
helloManager();
|
||||||
|
|
||||||
|
// Read the monitor's EDID (best-effort, when the device offers it) — the mode list a real
|
||||||
|
// driver derives from it; we log the preferred mode and keep our fixed offered list.
|
||||||
|
readEdid();
|
||||||
|
|
||||||
|
// Paint a known pattern, present it, and read it back — the V3 self-test that proves the
|
||||||
|
// whole path (virtqueue, resource, shared backing, transfer, flush) before a client attaches.
|
||||||
|
const pixels: [*]u32 = @ptrCast(@alignCast(surface.ptr));
|
||||||
|
const pixel_count: usize = @as(usize, max_width) * max_height;
|
||||||
|
for (0..pixel_count) |i| pixels[i] = testPixel(@intCast(i));
|
||||||
|
|
||||||
|
if (!presentFull()) {
|
||||||
|
log("virtio-gpu: initial present failed\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// The scanout surface is CPU-visible RAM: read the pattern back to prove the mapping,
|
||||||
|
// which together with the flush ack above is the automated stand-in for "it's on screen".
|
||||||
|
mmio.rmb();
|
||||||
|
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(@intCast(pixel_count / 2))) {
|
||||||
|
log("virtio-gpu: pixel read-back mismatch\n", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
log("virtio-gpu: flush acked, pixel check ok\n", .{});
|
||||||
|
|
||||||
|
// Offer the shared surface to the compositor so it upgrades off the GOP floor (V4).
|
||||||
|
announce();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Point scanout 0 at the current-mode rectangle of the resource. Reused by initial bring-up
|
||||||
|
/// and by `set_mode`.
|
||||||
|
fn setScanoutRect() bool {
|
||||||
|
const request = requestAt(vg.SetScanout);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.set_scanout) },
|
||||||
|
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||||
|
.scanout_id = 0,
|
||||||
|
.resource_id = resource_id,
|
||||||
|
};
|
||||||
|
return command_nodata(@sizeOf(vg.SetScanout)) == ok_nodata;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read and log the monitor's preferred mode from its EDID (VIRTIO_GPU_F_EDID). Best-effort:
|
||||||
|
/// a device that doesn't offer EDID, or a missing/short block, is logged and ignored.
|
||||||
|
fn readEdid() void {
|
||||||
|
if (!edid_available) {
|
||||||
|
log("virtio-gpu: EDID not offered by device\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const request = requestAt(vg.GetEdid);
|
||||||
|
request.* = .{ .hdr = .{ .type = @intFromEnum(vg.CmdType.get_edid) }, .scanout = 0 };
|
||||||
|
if (!submit(@sizeOf(vg.GetEdid), @sizeOf(vg.RespEdid))) {
|
||||||
|
log("virtio-gpu: EDID request not acked\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const response: *vg.RespEdid = @ptrFromInt(command.virtual + response_offset);
|
||||||
|
if (response.hdr.type != @intFromEnum(vg.CmdType.resp_ok_edid) or response.size < 64) {
|
||||||
|
log("virtio-gpu: EDID unavailable\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// The first detailed timing descriptor (EDID base-block offset 54) is the preferred mode:
|
||||||
|
// active pixels are 12-bit, low byte + high nibble (bytes 2/4 horizontal, 5/7 vertical).
|
||||||
|
const e = &response.edid;
|
||||||
|
const h_active = @as(u32, e[56]) | (@as(u32, e[58] & 0xF0) << 4);
|
||||||
|
const v_active = @as(u32, e[59]) | (@as(u32, e[61] & 0xF0) << 4);
|
||||||
|
log("virtio-gpu: EDID preferred mode {d}x{d}\n", .{ h_active, v_active });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Present the whole surface: copy the guest backing into the host resource, then flush it to
|
||||||
|
/// the panel. Reused by the V3 self-test and by every compositor present over `.scanout`. V4
|
||||||
|
/// presents the full surface; the damage-rect fast path is a later refinement.
|
||||||
|
fn presentFull() bool {
|
||||||
|
mmio.wmb(); // the surface writes must be visible before the device transfers them
|
||||||
|
{
|
||||||
|
// Transfer the current-mode rectangle from the guest backing to the host resource. The
|
||||||
|
// device uses the resource's (max) width as the row stride, so the top-left rect at
|
||||||
|
// offset 0 is exactly the visible area — the compositor composes at that same stride.
|
||||||
|
const request = requestAt(vg.TransferToHost2d);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.transfer_to_host_2d) },
|
||||||
|
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||||
|
.offset = 0,
|
||||||
|
.resource_id = resource_id,
|
||||||
|
};
|
||||||
|
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) return false;
|
||||||
|
}
|
||||||
|
{
|
||||||
|
// A fenced flush (vsync): the device signals the fence when the frame is actually on
|
||||||
|
// screen — which its used-ring ack, what our synchronous submit waits on, already gates.
|
||||||
|
const request = requestAt(vg.ResourceFlush);
|
||||||
|
request.* = .{
|
||||||
|
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_flush), .flags = vg.flag_fence, .fence_id = fence_next },
|
||||||
|
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||||
|
.resource_id = resource_id,
|
||||||
|
};
|
||||||
|
fence_next += 1;
|
||||||
|
if (command_nodata(@sizeOf(vg.ResourceFlush)) != ok_nodata) return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hello the device manager (role: bus — we own a PCI function, though we report no children):
|
||||||
|
/// the handshake that marks us up so the manager doesn't stop us at the hello deadline, and
|
||||||
|
/// (as a supervised driver) restarts us if we die. Best-effort: without a manager we still run.
|
||||||
|
fn helloManager() void {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
const manager = while (tries < 100) : (tries += 1) {
|
||||||
|
if (ipc.lookup(.device_manager)) |h| break h;
|
||||||
|
system.sleep(20);
|
||||||
|
} else {
|
||||||
|
log("virtio-gpu: no device manager to hello\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const hello = dm.Hello{ .role = @intFromEnum(dm.Role.bus), .device_id = device_id };
|
||||||
|
var reply: [dm.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch {
|
||||||
|
log("virtio-gpu: hello call failed\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (n < dm.reply_size or std.mem.bytesToValue(dm.HelloReply, reply[0..dm.reply_size]).status != 0) {
|
||||||
|
log("virtio-gpu: hello refused\n", .{});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
log("virtio-gpu: hello acknowledged\n", .{});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Announce the scanout to the display service so it upgrades off the GOP framebuffer: hand it
|
||||||
|
/// the shared surface as a capability plus the geometry. Best-effort and non-fatal — without a
|
||||||
|
/// display service (the standalone virtio-gpu bring-up test) the driver is still a valid
|
||||||
|
/// scanout service; it just serves no one. The display replies immediately (it defers its
|
||||||
|
/// first present to a timer), so this returns before we start serving `.scanout` — no deadlock.
|
||||||
|
fn announce() void {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
const display = while (tries < 50) : (tries += 1) {
|
||||||
|
if (ipc.lookup(.display)) |h| break h;
|
||||||
|
system.sleep(20);
|
||||||
|
} else {
|
||||||
|
log("virtio-gpu: no display service to announce to (scanout-only)\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var request = dp.Request{
|
||||||
|
.operation = @intFromEnum(dp.Operation.attach_scanout),
|
||||||
|
.x = max_width, // the shared surface's row stride in pixels (it is sized to the max mode)
|
||||||
|
.width = current_width,
|
||||||
|
.height = current_height,
|
||||||
|
.colour = display_format_bgrx,
|
||||||
|
};
|
||||||
|
var reply: [dp.reply_size]u8 = undefined;
|
||||||
|
_ = ipc.callCap(display, std.mem.asBytes(&request), &reply, surface.handle) catch {
|
||||||
|
log("virtio-gpu: announce to display failed\n", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
log("virtio-gpu: announced scanout to display\n", .{});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A `sp.Reply{status}` written into `reply`.
|
||||||
|
fn scanoutStatus(reply: []u8, ok: bool) usize {
|
||||||
|
const response = sp.Reply{ .status = if (ok) 0 else -1 };
|
||||||
|
@memcpy(reply[0..sp.reply_size], std.mem.asBytes(&response));
|
||||||
|
return sp.reply_size;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
|
||||||
|
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
|
||||||
|
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
if (message.len < sp.request_size) return 0;
|
||||||
|
const request = std.mem.bytesToValue(sp.Request, message[0..sp.request_size]);
|
||||||
|
switch (request.operation) {
|
||||||
|
@intFromEnum(sp.Operation.present) => return scanoutStatus(reply, presentFull()),
|
||||||
|
@intFromEnum(sp.Operation.get_modes) => {
|
||||||
|
var response = sp.ModesReply{ .status = 0, .count = offered_modes.len, .modes = undefined };
|
||||||
|
for (0..sp.max_modes) |i| {
|
||||||
|
response.modes[i] = if (i < offered_modes.len)
|
||||||
|
.{ .width = offered_modes[i].width, .height = offered_modes[i].height }
|
||||||
|
else
|
||||||
|
.{ .width = 0, .height = 0 };
|
||||||
|
}
|
||||||
|
@memcpy(reply[0..sp.modes_reply_size], std.mem.asBytes(&response));
|
||||||
|
return sp.modes_reply_size;
|
||||||
|
},
|
||||||
|
@intFromEnum(sp.Operation.set_mode) => {
|
||||||
|
const w = request.width;
|
||||||
|
const h = request.height;
|
||||||
|
if (w == 0 or h == 0 or w > max_width or h > max_height) return scanoutStatus(reply, false);
|
||||||
|
current_width = w;
|
||||||
|
current_height = h;
|
||||||
|
return scanoutStatus(reply, setScanoutRect());
|
||||||
|
},
|
||||||
|
else => return 0,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse {
|
||||||
|
_ = system.write("virtio-gpu: missing device id (argv[1])\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
log("virtio-gpu: malformed device id '{s}'\n", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
runtime.service.run(256, .{
|
||||||
|
.service = .scanout,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
//! virtio 1.0 PCI transport — the vendor capabilities in PCI config space that point at the
|
||||||
|
//! device's structures (common config, notify, ISR) in a BAR, the common-config register
|
||||||
|
//! block, and the split-virtqueue layout. `extern` structs matching the spec. Host-tested
|
||||||
|
//! for size. See docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// PCI vendor-specific capability id (0x09) — virtio 1.0 structures are advertised as these.
|
||||||
|
pub const pci_cap_vendor: u8 = 0x09;
|
||||||
|
|
||||||
|
/// virtio_pci_cap `cfg_type`: which structure a vendor capability points at.
|
||||||
|
pub const cfg_common: u8 = 1;
|
||||||
|
pub const cfg_notify: u8 = 2;
|
||||||
|
pub const cfg_isr: u8 = 3;
|
||||||
|
pub const cfg_device: u8 = 4;
|
||||||
|
pub const cfg_pci: u8 = 5;
|
||||||
|
|
||||||
|
/// virtio_pci_cap — a vendor capability naming a structure at (bar, offset, length) within
|
||||||
|
/// a PCI BAR. Read straight out of config space.
|
||||||
|
pub const PciCap = extern struct {
|
||||||
|
cap_vndr: u8, // 0x09
|
||||||
|
cap_next: u8, // next capability's offset in config space (0 = end)
|
||||||
|
cap_len: u8,
|
||||||
|
cfg_type: u8, // cfg_common / cfg_notify / ...
|
||||||
|
bar: u8, // which BAR the structure lives in
|
||||||
|
padding: [3]u8,
|
||||||
|
offset: u32, // offset within the BAR
|
||||||
|
length: u32, // length of the structure
|
||||||
|
};
|
||||||
|
|
||||||
|
/// virtio_pci_notify_cap: a notify capability carries a multiplier after the base cap; the
|
||||||
|
/// per-queue notify address is `notify_base + queue_notify_off * notify_off_multiplier`.
|
||||||
|
pub const NotifyCap = extern struct {
|
||||||
|
cap: PciCap,
|
||||||
|
notify_off_multiplier: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// virtio_pci_common_cfg — the common configuration register block (little-endian MMIO).
|
||||||
|
pub const CommonCfg = extern struct {
|
||||||
|
device_feature_select: u32,
|
||||||
|
device_feature: u32,
|
||||||
|
driver_feature_select: u32,
|
||||||
|
driver_feature: u32,
|
||||||
|
msix_config: u16,
|
||||||
|
num_queues: u16,
|
||||||
|
device_status: u8,
|
||||||
|
config_generation: u8,
|
||||||
|
queue_select: u16,
|
||||||
|
queue_size: u16,
|
||||||
|
queue_msix_vector: u16,
|
||||||
|
queue_enable: u16,
|
||||||
|
queue_notify_off: u16,
|
||||||
|
queue_desc: u64,
|
||||||
|
queue_driver: u64,
|
||||||
|
queue_device: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// device_status bits (written to `CommonCfg.device_status` during bring-up).
|
||||||
|
pub const status_acknowledge: u8 = 1;
|
||||||
|
pub const status_driver: u8 = 2;
|
||||||
|
pub const status_driver_ok: u8 = 4;
|
||||||
|
pub const status_features_ok: u8 = 8;
|
||||||
|
|
||||||
|
/// VIRTIO_F_VERSION_1 — feature bit 32 (in the second 32-bit feature word). Required for a
|
||||||
|
/// modern device; we negotiate exactly this bit and nothing else.
|
||||||
|
pub const feature_version_1_word: u32 = 1; // device_feature_select value for bits 32..63
|
||||||
|
pub const feature_version_1_bit: u32 = 1 << 0; // bit 32 within that word
|
||||||
|
|
||||||
|
// --- split virtqueue -------------------------------------------------------
|
||||||
|
|
||||||
|
pub const Desc = extern struct {
|
||||||
|
addr: u64, // guest-physical
|
||||||
|
len: u32,
|
||||||
|
flags: u16,
|
||||||
|
next: u16,
|
||||||
|
};
|
||||||
|
pub const desc_flag_next: u16 = 1; // buffer continues in `next`
|
||||||
|
pub const desc_flag_write: u16 = 2; // device-writable (else driver-writable/device-readable)
|
||||||
|
|
||||||
|
/// The available ring's fixed header; a `[queue_size]u16` ring and a trailing `used_event`
|
||||||
|
/// u16 follow it in memory (laid out by the driver).
|
||||||
|
pub const AvailHdr = extern struct {
|
||||||
|
flags: u16,
|
||||||
|
idx: u16,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One entry of the used ring.
|
||||||
|
pub const UsedElem = extern struct {
|
||||||
|
id: u32,
|
||||||
|
len: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The used ring's fixed header; a `[queue_size]UsedElem` ring and a trailing `avail_event`
|
||||||
|
/// u16 follow it.
|
||||||
|
pub const UsedHdr = extern struct {
|
||||||
|
flags: u16,
|
||||||
|
idx: u16,
|
||||||
|
};
|
||||||
|
|
||||||
|
test "virtio-pci struct sizes match the spec" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(PciCap));
|
||||||
|
try std.testing.expectEqual(@as(usize, 56), @sizeOf(CommonCfg));
|
||||||
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Desc));
|
||||||
|
try std.testing.expectEqual(@as(usize, 8), @sizeOf(UsedElem));
|
||||||
|
}
|
||||||
@@ -80,6 +80,35 @@ var timer_hz: u32 = 0;
|
|||||||
var tsc_hz: u64 = 0;
|
var tsc_hz: u64 = 0;
|
||||||
var tsc_base: u64 = 0;
|
var tsc_base: u64 = 0;
|
||||||
|
|
||||||
|
/// Whether the TSC is architecturally **invariant** — a constant rate regardless of
|
||||||
|
/// P/C-state transitions, and thus valid as a clocksource (CPUID leaf 0x80000007,
|
||||||
|
/// EDX bit 8). AMD and modern Intel set it; the bare qemu64 model does not. Measured
|
||||||
|
/// frequency alone is not enough: a non-invariant TSC speeds up and slows down with
|
||||||
|
/// the core clock, so reading it as wall time would drift.
|
||||||
|
var tsc_invariant: bool = false;
|
||||||
|
/// Cleared if the cross-core warp check (checkWarpSource) ever sees the TSC read
|
||||||
|
/// lower on one core than the max another core has already published — i.e. the
|
||||||
|
/// per-core TSCs are not synchronized, and a task migrating cores could see time go
|
||||||
|
/// backward. Starts true (assume synchronized until proven otherwise).
|
||||||
|
var tsc_synced: bool = true;
|
||||||
|
/// The worst backward skew the warp check observed, in TSC cycles (0 = none).
|
||||||
|
var tsc_warp_cycles: u64 = 0;
|
||||||
|
|
||||||
|
/// The monotonic clock's source. The TSC when it is invariant *and* synchronized —
|
||||||
|
/// the fast `rdtsc` path taken on real Intel/AMD and modern VMs. Otherwise the HPET
|
||||||
|
/// main counter: a single fixed-rate counter, immune to both per-core skew and
|
||||||
|
/// frequency scaling, so it stays accurate on a bare VM or a warped machine.
|
||||||
|
const ClockSource = enum { tsc, hpet };
|
||||||
|
var clock_source: ClockSource = .tsc;
|
||||||
|
|
||||||
|
/// HPET standby clocksource, set up in calibrate() whenever an HPET exists (whether
|
||||||
|
/// or not calibration itself measured against it): its frequency, the counter value
|
||||||
|
/// chosen as the zero point, and its width mask. Only a 64-bit HPET is used as a
|
||||||
|
/// clocksource — a 32-bit one wraps too fast to be monotonic without accumulation.
|
||||||
|
var hpet_clock_hz: u64 = 0;
|
||||||
|
var hpet_clock_base: u64 = 0;
|
||||||
|
var hpet_clock_mask: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
/// Read the 64-bit Time Stamp Counter.
|
/// Read the 64-bit Time Stamp Counter.
|
||||||
fn rdtsc() u64 {
|
fn rdtsc() u64 {
|
||||||
var low: u32 = undefined;
|
var low: u32 = undefined;
|
||||||
@@ -220,6 +249,31 @@ pub fn calibrate() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||||
|
|
||||||
|
// Decide whether the TSC is trustworthy as a clocksource. Frequency (measured
|
||||||
|
// above, possibly against the HPET/PIT) is necessary but not sufficient: the TSC
|
||||||
|
// must also be *invariant* (CPUID 0x80000007 EDX[8]). AMD and modern Intel set
|
||||||
|
// this; the bare qemu64 model does not.
|
||||||
|
tsc_invariant = tscIsInvariant();
|
||||||
|
|
||||||
|
// Bring up the HPET as a standby clocksource whenever one exists — even on the
|
||||||
|
// CPUID-0x15 path where calibration never touched it — so a non-invariant TSC
|
||||||
|
// (here) or an unsynchronized one (checkWarpSource, during SMP bring-up) can fall
|
||||||
|
// back to a source that is immune to both. hpetHz() maps + enables the counter
|
||||||
|
// and is idempotent if calibration already used it.
|
||||||
|
if (configuration_hpet_base != 0) {
|
||||||
|
if (hpetHz()) |hz| {
|
||||||
|
hpet_clock_mask = hpetMask();
|
||||||
|
if (hpet_clock_mask == ~@as(u64, 0)) { // only a 64-bit HPET is monotonic enough
|
||||||
|
hpet_clock_hz = hz;
|
||||||
|
hpet_clock_base = readHpet();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Select the source: the fast TSC when invariant, else the HPET if we have one.
|
||||||
|
// (checkWarpSource may still demote TSC -> HPET later if the cores' TSCs skew.)
|
||||||
|
if (!tsc_invariant and hpet_clock_hz != 0) clock_source = .hpet;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||||
@@ -275,7 +329,7 @@ fn calibratePit() void {
|
|||||||
// --- reference clocks ------------------------------------------------------
|
// --- reference clocks ------------------------------------------------------
|
||||||
|
|
||||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
/// null if the CPU doesn't enumerate it (common under QEMU, and on AMD).
|
||||||
fn cpuidTscHz() ?u64 {
|
fn cpuidTscHz() ?u64 {
|
||||||
if (cpuid(0).eax < 0x15) return null;
|
if (cpuid(0).eax < 0x15) return null;
|
||||||
const r = cpuid(0x15);
|
const r = cpuid(0x15);
|
||||||
@@ -283,6 +337,15 @@ fn cpuidTscHz() ?u64 {
|
|||||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the CPU advertises an **invariant** TSC (CPUID leaf 0x80000007, EDX
|
||||||
|
/// bit 8) — the architectural guarantee, on both Intel and AMD, that the TSC ticks
|
||||||
|
/// at a constant rate across P/C-states and never stops. Requires the extended-leaf
|
||||||
|
/// range to reach 0x80000007 first.
|
||||||
|
fn tscIsInvariant() bool {
|
||||||
|
if (cpuid(0x80000000).eax < 0x80000007) return false;
|
||||||
|
return (cpuid(0x80000007).edx & (1 << 8)) != 0;
|
||||||
|
}
|
||||||
|
|
||||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||||
|
|
||||||
fn cpuid(leaf: u32) CpuidRegs {
|
fn cpuid(leaf: u32) CpuidRegs {
|
||||||
@@ -310,11 +373,20 @@ fn hpetWrite64(off: usize, value: u64) void {
|
|||||||
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the HPET has been mapped into the physmap yet, so `configuration_hpet_base`
|
||||||
|
/// already holds the virtual address. `hpetHz` is called more than once (calibration
|
||||||
|
/// may use the HPET, and the standby-clocksource setup asks for it again), and mapping
|
||||||
|
/// an already-mapped base a second time would double-offset it into an overflow.
|
||||||
|
var hpet_mapped: bool = false;
|
||||||
|
|
||||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||||
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||||
/// address, so the register accessors reach it without the identity map.
|
/// address, so the register accessors reach it without the identity map. Idempotent.
|
||||||
fn hpetHz() ?u64 {
|
fn hpetHz() ?u64 {
|
||||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
if (!hpet_mapped) {
|
||||||
|
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||||
|
hpet_mapped = true;
|
||||||
|
}
|
||||||
const caps = hpetRead64(0x00);
|
const caps = hpetRead64(0x00);
|
||||||
const period_fs = caps >> 32; // femtoseconds per tick
|
const period_fs = caps >> 32; // femtoseconds per tick
|
||||||
if (period_fs == 0) return null;
|
if (period_fs == 0) return null;
|
||||||
@@ -364,24 +436,169 @@ pub fn tscHz() u64 {
|
|||||||
return tsc_hz;
|
return tsc_hz;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
// Monotonic high-resolution clock. A function per resolution, each scaling the
|
||||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
// counter delta directly at its unit (the 128-bit intermediate avoids overflow
|
||||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
// across a long uptime). nanos() resolves to a few ns on the TSC; millis() is what
|
||||||
// the scheduler uses for sleep deadlines.
|
// the scheduler uses for sleep deadlines. The source is the TSC when it is invariant
|
||||||
|
// and synchronized, else the HPET counter (see clock_source) — the branch is one
|
||||||
|
// global load and the TSC path is unchanged from before.
|
||||||
|
|
||||||
|
/// The selected source's counter delta since its zero point.
|
||||||
|
fn clockCount() u64 {
|
||||||
|
return switch (clock_source) {
|
||||||
|
.tsc => rdtsc() -% tsc_base,
|
||||||
|
// A 64-bit HPET (the only kind we select) never wraps in any realistic
|
||||||
|
// uptime, so the wrapping subtraction is exact.
|
||||||
|
.hpet => readHpet() -% hpet_clock_base,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The selected source's frequency (0 if the clock is unavailable/uncalibrated).
|
||||||
|
fn clockHertz() u64 {
|
||||||
|
return switch (clock_source) {
|
||||||
|
.tsc => tsc_hz,
|
||||||
|
.hpet => hpet_clock_hz,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
pub fn nanos() u64 {
|
pub fn nanos() u64 {
|
||||||
if (tsc_hz == 0) return 0;
|
const hz = clockHertz();
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
if (hz == 0) return 0;
|
||||||
|
return @intCast(@as(u128, clockCount()) * 1_000_000_000 / hz);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn micros() u64 {
|
pub fn micros() u64 {
|
||||||
if (tsc_hz == 0) return 0;
|
const hz = clockHertz();
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
if (hz == 0) return 0;
|
||||||
|
return @intCast(@as(u128, clockCount()) * 1_000_000 / hz);
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn millis() u64 {
|
pub fn millis() u64 {
|
||||||
if (tsc_hz == 0) return 0;
|
const hz = clockHertz();
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
if (hz == 0) return 0;
|
||||||
|
return @intCast(@as(u128, clockCount()) * 1_000 / hz);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the CPU advertises an invariant TSC (CPUID 0x80000007 EDX[8]).
|
||||||
|
pub fn tscInvariant() bool {
|
||||||
|
return tsc_invariant;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test hook: force the TSC clocksource on, as if the CPU had advertised an invariant
|
||||||
|
/// TSC. QEMU's TCG accelerator (the only one for an x86 guest on an Apple-Silicon
|
||||||
|
/// host) does not expose the invariant-TSC bit — its emulated TSC isn't invariant — so
|
||||||
|
/// the tsc-sync test can't reach the real-Intel/AMD/KVM path through CPUID. This lets
|
||||||
|
/// that test exercise the TSC clocksource and the cross-core warp check anyway. tsc_base
|
||||||
|
/// is left as-is so the switch from the HPET is continuous.
|
||||||
|
pub fn forceTscClocksourceForTest() void {
|
||||||
|
tsc_invariant = true;
|
||||||
|
clock_source = .tsc;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How many per-AP warp checks actually ran (a rendezvous completed) — lets a test
|
||||||
|
/// confirm the cross-core check executed rather than being skipped.
|
||||||
|
pub fn warpChecksRun() u32 {
|
||||||
|
return warp_checks;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the per-core TSCs are synchronized (no backward warp seen at bring-up).
|
||||||
|
pub fn tscSynced() bool {
|
||||||
|
return tsc_synced;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The active monotonic clocksource, for the boot log and tests.
|
||||||
|
pub fn clockSourceName() []const u8 {
|
||||||
|
return switch (clock_source) {
|
||||||
|
.tsc => "tsc",
|
||||||
|
.hpet => "hpet",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- cross-core TSC synchronization ("warp") check -------------------------
|
||||||
|
// Two cores hammer a shared "max seen" TSC value under a lock; if either reads a
|
||||||
|
// value below that max, its TSC lags the other's, and time would run backward for a
|
||||||
|
// task migrating between them (Linux calls this a warp). danos brings APs up one at a
|
||||||
|
// time, so this runs pairwise: the BSP (source) against each AP (target) as it comes
|
||||||
|
// online. It only matters — and only runs — while the TSC is the clocksource; on a
|
||||||
|
// machine already on the HPET (a bare VM) the whole rendezvous is skipped.
|
||||||
|
|
||||||
|
var warp_lock: u32 = 0;
|
||||||
|
var warp_last: u64 = 0;
|
||||||
|
var warp_bsp_ready: u32 = 0;
|
||||||
|
var warp_ap_ready: u32 = 0;
|
||||||
|
var warp_stop: u32 = 0;
|
||||||
|
var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test)
|
||||||
|
|
||||||
|
const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates
|
||||||
|
const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot
|
||||||
|
|
||||||
|
fn warpTick() void {
|
||||||
|
while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause");
|
||||||
|
const t = rdtsc();
|
||||||
|
if (t < warp_last) {
|
||||||
|
const delta = warp_last - t;
|
||||||
|
if (delta > tsc_warp_cycles) tsc_warp_cycles = delta;
|
||||||
|
tsc_synced = false;
|
||||||
|
} else {
|
||||||
|
warp_last = t;
|
||||||
|
}
|
||||||
|
@atomicStore(u32, &warp_lock, 0, .release);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spin (bounded) until `flag` is nonzero; false on timeout.
|
||||||
|
fn warpAwait(flag: *u32) bool {
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) {
|
||||||
|
if (spins >= warp_spin_limit) return false;
|
||||||
|
asm volatile ("pause");
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// BSP side of the pairwise TSC warp check, run once per AP as it reports in. No-op
|
||||||
|
/// unless the TSC is the active clocksource. If the AP's TSC proves to lag, demote
|
||||||
|
/// the monotonic clock to the HPET without a discontinuity.
|
||||||
|
pub fn checkWarpSource() void {
|
||||||
|
if (clock_source != .tsc) return;
|
||||||
|
warp_last = 0;
|
||||||
|
@atomicStore(u32, &warp_stop, 0, .release);
|
||||||
|
@atomicStore(u32, &warp_ap_ready, 0, .release);
|
||||||
|
@atomicStore(u32, &warp_bsp_ready, 1, .release);
|
||||||
|
if (!warpAwait(&warp_ap_ready)) { // AP never joined the rendezvous; skip, don't hang
|
||||||
|
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < warp_rounds) : (i += 1) warpTick();
|
||||||
|
@atomicStore(u32, &warp_stop, 1, .release);
|
||||||
|
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||||
|
warp_checks += 1;
|
||||||
|
|
||||||
|
if (!tsc_synced and hpet_clock_hz != 0) demoteToHpet();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// AP side: join the BSP's warp check, then return so the core can enter the
|
||||||
|
/// scheduler. Bounded so a missing BSP can't strand the core.
|
||||||
|
pub fn checkWarpTarget() void {
|
||||||
|
if (clock_source != .tsc) return;
|
||||||
|
if (!warpAwait(&warp_bsp_ready)) return;
|
||||||
|
@atomicStore(u32, &warp_ap_ready, 1, .release);
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) {
|
||||||
|
if (spins >= warp_spin_limit) return;
|
||||||
|
warpTick();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Switch the clocksource from the TSC to the HPET without a discontinuity: choose
|
||||||
|
/// the HPET zero point so it reads the same nanosecond value the TSC does right now,
|
||||||
|
/// so time neither jumps nor runs backward across the switch. Called when the warp
|
||||||
|
/// check proves the per-core TSCs unsynchronized.
|
||||||
|
fn demoteToHpet() void {
|
||||||
|
const now_ns = @as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz;
|
||||||
|
const equivalent_ticks: u64 = @intCast(now_ns * hpet_clock_hz / 1_000_000_000);
|
||||||
|
hpet_clock_base = readHpet() -% equivalent_ticks;
|
||||||
|
clock_source = .hpet;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||||
|
|||||||
@@ -106,6 +106,12 @@ pub fn serialWrite(bytes: []const u8) void {
|
|||||||
serial.write(bytes);
|
serial.write(bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether a working UART was detected (loopback probe). When false the serial
|
||||||
|
/// sink is silently inert — a dead legacy COM1 costs nothing per byte.
|
||||||
|
pub fn serialPresent() bool {
|
||||||
|
return serial.present();
|
||||||
|
}
|
||||||
|
|
||||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||||
/// displays. The last-resort progress signal when there's no text output at all.
|
/// displays. The last-resort progress signal when there's no text output at all.
|
||||||
@@ -162,10 +168,18 @@ pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, e
|
|||||||
paging.mapUserInto(root, virtual, physical, writable, executable);
|
paging.mapUserInto(root, virtual, physical, writable, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
/// Map a device MMIO window into address space `root`: RW+NX, and marked so teardown
|
||||||
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
/// won't free the MMIO frames as RAM. `write_combining` picks the cache type —
|
||||||
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
/// false = strong-uncacheable (registers), true = write-combining (a framebuffer).
|
||||||
paging.mapUserDeviceInto(root, virtual, physical, len);
|
/// For IO passthrough.
|
||||||
|
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64, write_combining: bool) void {
|
||||||
|
paging.mapUserDeviceInto(root, virtual, physical, len, write_combining);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is the user leaf mapping `virtual` in address space `root` write-combining? Null if
|
||||||
|
/// unmapped. For tests verifying the framebuffer map's cache type.
|
||||||
|
pub fn userLeafIsWriteCombining(root: u64, virtual: u64) ?bool {
|
||||||
|
return paging.leafIsWriteCombining(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map coherent DMA RAM into address space `root`: strong-uncacheable, RW+NX, but
|
/// Map coherent DMA RAM into address space `root`: strong-uncacheable, RW+NX, but
|
||||||
@@ -174,6 +188,13 @@ pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
|||||||
paging.mapUserDmaInto(root, virtual, physical, len);
|
paging.mapUserDmaInto(root, virtual, physical, len);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map shared cacheable RAM into address space `root`: write-back cacheable, RW+NX, and
|
||||||
|
/// marked so teardown won't free the frames (they're owned by a refcounted shm object,
|
||||||
|
/// freed when its last capability drops). For shm_create/shm_map.
|
||||||
|
pub fn mapUserSharedInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||||
|
paging.mapUserSharedInto(root, virtual, physical, len);
|
||||||
|
}
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||||
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||||
paging.map(virtual, physical, writable);
|
paging.map(virtual, physical, writable);
|
||||||
@@ -469,6 +490,124 @@ pub fn clockHz() u64 {
|
|||||||
return apic.tscHz();
|
return apic.tscHz();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- real-time clock (CMOS) --------------------------------------------------
|
||||||
|
//
|
||||||
|
// The battery-backed CMOS clock, read once at boot and thereafter anchored to the
|
||||||
|
// monotonic clock (see kernel/wall-clock.zig) — so this is never on a hot path and
|
||||||
|
// needs no lock. Wall-clock *seconds* are mechanism the kernel owns (the hardware's
|
||||||
|
// value), like the monotonic clock; calendars/timezones are policy layered on top.
|
||||||
|
|
||||||
|
fn cmosRead(register: u8) u8 {
|
||||||
|
io.outb(0x70, register);
|
||||||
|
return io.inb(0x71);
|
||||||
|
}
|
||||||
|
|
||||||
|
const RtcFields = struct { second: u8, minute: u8, hour: u8, day: u8, month: u8, year: u8 };
|
||||||
|
|
||||||
|
fn rtcRaw() RtcFields {
|
||||||
|
while (cmosRead(0x0A) & 0x80 != 0) {} // wait out any update in progress (status A bit 7)
|
||||||
|
return .{
|
||||||
|
.second = cmosRead(0x00),
|
||||||
|
.minute = cmosRead(0x02),
|
||||||
|
.hour = cmosRead(0x04),
|
||||||
|
.day = cmosRead(0x07),
|
||||||
|
.month = cmosRead(0x08),
|
||||||
|
.year = cmosRead(0x09),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bcdToBinary(v: u8) u8 {
|
||||||
|
return (v & 0x0F) + ((v >> 4) * 10);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn isLeapYear(y: u32) bool {
|
||||||
|
return (y % 4 == 0 and y % 100 != 0) or (y % 400 == 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read the CMOS real-time clock and convert it to Unix epoch seconds (UTC).
|
||||||
|
pub fn readRtcUnixSeconds() u64 {
|
||||||
|
// Read until two consecutive reads agree, so we never latch a half-updated time.
|
||||||
|
var a = rtcRaw();
|
||||||
|
while (true) {
|
||||||
|
const b = rtcRaw();
|
||||||
|
if (a.second == b.second and a.minute == b.minute and a.hour == b.hour and
|
||||||
|
a.day == b.day and a.month == b.month and a.year == b.year) break;
|
||||||
|
a = b;
|
||||||
|
}
|
||||||
|
|
||||||
|
const status_b = cmosRead(0x0B);
|
||||||
|
const binary_mode = status_b & 0x04 != 0; // else BCD
|
||||||
|
const hour_24 = status_b & 0x02 != 0; // else 12-hour with a PM bit
|
||||||
|
|
||||||
|
var second = a.second;
|
||||||
|
var minute = a.minute;
|
||||||
|
var hour_field = a.hour;
|
||||||
|
var day = a.day;
|
||||||
|
var month = a.month;
|
||||||
|
var year = a.year;
|
||||||
|
if (!binary_mode) {
|
||||||
|
second = bcdToBinary(second);
|
||||||
|
minute = bcdToBinary(minute);
|
||||||
|
hour_field = bcdToBinary(hour_field & 0x7F) | (hour_field & 0x80); // preserve the PM bit
|
||||||
|
day = bcdToBinary(day);
|
||||||
|
month = bcdToBinary(month);
|
||||||
|
year = bcdToBinary(year);
|
||||||
|
}
|
||||||
|
|
||||||
|
var hour: u32 = hour_field & 0x7F;
|
||||||
|
if (!hour_24) {
|
||||||
|
const pm = hour_field & 0x80 != 0;
|
||||||
|
hour %= 12; // 12 AM/PM -> 0
|
||||||
|
if (pm) hour += 12;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The CMOS year is 0..99; QEMU and modern hardware mean 20xx (there is no
|
||||||
|
// reliable century register on QEMU). Treat < 70 as 20xx, else 19xx.
|
||||||
|
const full_year: u32 = if (year < 70) 2000 + @as(u32, year) else 1900 + @as(u32, year);
|
||||||
|
|
||||||
|
var days: u64 = 0;
|
||||||
|
var y: u32 = 1970;
|
||||||
|
while (y < full_year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||||
|
const month_lengths = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||||
|
var m: u8 = 1;
|
||||||
|
while (m < month) : (m += 1) {
|
||||||
|
days += month_lengths[m - 1];
|
||||||
|
if (m == 2 and isLeapYear(full_year)) days += 1;
|
||||||
|
}
|
||||||
|
days += @as(u64, day) - 1;
|
||||||
|
|
||||||
|
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the CPU guarantees an **invariant** TSC (CPUID 0x80000007 EDX[8] on
|
||||||
|
/// x86; the analogous architectural guarantee elsewhere). When false the TSC is not
|
||||||
|
/// used as the clocksource.
|
||||||
|
pub fn clockInvariant() bool {
|
||||||
|
return apic.tscInvariant();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the per-core clock counters are synchronized (no backward warp observed
|
||||||
|
/// at SMP bring-up). When false the clock falls back off the TSC.
|
||||||
|
pub fn clockSynchronized() bool {
|
||||||
|
return apic.tscSynced();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The active monotonic clocksource, for the boot log ("tsc" or "hpet" on x86).
|
||||||
|
pub fn clockSourceName() []const u8 {
|
||||||
|
return apic.clockSourceName();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test hook: force the TSC clocksource on, to exercise the TSC + warp-check path on
|
||||||
|
/// a hypervisor that won't advertise an invariant TSC (see apic.forceTscClocksourceForTest).
|
||||||
|
pub fn forceTscClocksourceForTest() void {
|
||||||
|
apic.forceTscClocksourceForTest();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How many per-AP TSC warp checks completed (for the tsc-sync test).
|
||||||
|
pub fn warpChecksRun() u32 {
|
||||||
|
return apic.warpChecksRun();
|
||||||
|
}
|
||||||
|
|
||||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||||
pub fn enableInterrupts() void {
|
pub fn enableInterrupts() void {
|
||||||
asm volatile ("sti");
|
asm volatile ("sti");
|
||||||
|
|||||||
@@ -205,7 +205,17 @@ syscall_entry:
|
|||||||
push %r14
|
push %r14
|
||||||
push %r15
|
push %r15
|
||||||
mov %rsp, %rdi # trap-frame pointer
|
mov %rsp, %rdi # trap-frame pointer
|
||||||
|
# Preserve the caller's SSE/x87 register file across the syscall — see the same
|
||||||
|
# dance in isr_common. Without it a syscall (or a task the scheduler runs while
|
||||||
|
# this one blocks) clobbers the caller's live XMM values, which the compiler is
|
||||||
|
# free to hold across a syscall (its wrappers only clobber rcx/r11/memory).
|
||||||
|
mov %rsp, %rbx
|
||||||
|
and $-16, %rsp
|
||||||
|
sub $512, %rsp
|
||||||
|
fxsave (%rsp)
|
||||||
call interruptDispatch
|
call interruptDispatch
|
||||||
|
fxrstor (%rsp)
|
||||||
|
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
||||||
pop %r15
|
pop %r15
|
||||||
pop %r14
|
pop %r14
|
||||||
pop %r13
|
pop %r13
|
||||||
@@ -357,7 +367,21 @@ isr_common:
|
|||||||
push %r14
|
push %r14
|
||||||
push %r15
|
push %r15
|
||||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||||
|
# Save the interrupted SSE/x87 register file before any kernel code runs, and
|
||||||
|
# restore it on the way out — the kernel and user both keep live values in XMM
|
||||||
|
# (a 16-byte struct copy is a movdqu), and the kernel never otherwise preserves
|
||||||
|
# them, so an interrupt handler (and whatever the scheduler runs in its place)
|
||||||
|
# would silently clobber the interrupted task's vector registers. rbx bridges the
|
||||||
|
# exact rsp across the call: it is callee-saved (interruptDispatch and every
|
||||||
|
# context switch preserve it), so it survives even a blocking dispatch, and the
|
||||||
|
# `and`/`sub` gives fxsave its required 16-byte-aligned scratch on the kernel stack.
|
||||||
|
mov %rsp, %rbx
|
||||||
|
and $-16, %rsp
|
||||||
|
sub $512, %rsp
|
||||||
|
fxsave (%rsp)
|
||||||
call interruptDispatch
|
call interruptDispatch
|
||||||
|
fxrstor (%rsp)
|
||||||
|
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
||||||
pop %r15
|
pop %r15
|
||||||
pop %r14
|
pop %r14
|
||||||
pop %r13
|
pop %r13
|
||||||
|
|||||||
@@ -7,8 +7,13 @@
|
|||||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||||
//! which the kernel heap will build on.
|
//! which the kernel heap will build on.
|
||||||
//!
|
//!
|
||||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
//! The physmap (the permanent window onto all physical RAM) is built with 2 MiB
|
||||||
//! negligible against available RAM.
|
//! huge pages wherever the range is 2 MiB-aligned, falling back to 4 KiB for the
|
||||||
|
//! unaligned edges. On a big machine that is the difference between ~16.7M page-
|
||||||
|
//! table entries (128 MiB of tables) and ~32K — it makes both the build and the
|
||||||
|
//! footprint scale sanely with RAM. Everything else (kernel segments, heap, user
|
||||||
|
//! space, on-demand MMIO) stays 4 KiB: precise, and the table memory is
|
||||||
|
//! negligible there.
|
||||||
|
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
@@ -22,10 +27,22 @@ const writable: u64 = 1 << 1;
|
|||||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||||
const pwt: u64 = 1 << 3; // page write-through
|
const pwt: u64 = 1 << 3; // page write-through
|
||||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||||
|
const page_size_bit: u64 = 1 << 7; // PS: this PDPT/PD entry is a 1 GiB/2 MiB leaf, not a pointer to the next table
|
||||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||||
const no_execute: u64 = 1 << 63;
|
const no_execute: u64 = 1 << 63;
|
||||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
|
// The PAT-index bit. In a 4 KiB PTE it is bit 7; in a huge leaf (2 MiB PDE / 1 GiB
|
||||||
|
// PDPTE) bit 7 is PS, so the PAT bit moves to bit 12. With PCD=PWT=0 this selects
|
||||||
|
// PAT entry 4, which `setupPat` programs to write-combining (see mapRangePhysmap).
|
||||||
|
const pte_pat: u64 = 1 << 7;
|
||||||
|
const huge_pat: u64 = 1 << 12;
|
||||||
|
const ia32_pat: u32 = 0x277;
|
||||||
|
|
||||||
|
/// The physmap's page size for 2 MiB-aligned RAM: one PD leaf covers this instead
|
||||||
|
/// of 512 PT entries. 4 KiB pages fill the unaligned edges (see mapRangePhysmap).
|
||||||
|
const huge_page_size: u64 = 2 << 20; // 2 MiB
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
// ELF segment flags (p_flags).
|
||||||
const pf_x: u32 = 1;
|
const pf_x: u32 = 1;
|
||||||
const pf_w: u32 = 2;
|
const pf_w: u32 = 2;
|
||||||
@@ -74,7 +91,15 @@ fn allocTable() u64 {
|
|||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||||
/// writable only if every level is; non-executable if any level is).
|
/// writable only if every level is; non-executable if any level is).
|
||||||
fn descend(entry: *u64) u64 {
|
fn descend(entry: *u64) u64 {
|
||||||
if (entry.* & present != 0) return entry.* & address_mask;
|
if (entry.* & present != 0) {
|
||||||
|
// A present-but-huge entry is a leaf, not a table: descending would read
|
||||||
|
// its 2 MiB/1 GiB data frame as a page table and corrupt RAM. This only
|
||||||
|
// fires on a bug — a 4 KiB map landing inside a physmap huge page — and a
|
||||||
|
// loud panic beats silent corruption. (The physmap and the 4 KiB regions
|
||||||
|
// live in disjoint PML4 slots, so it should never happen.)
|
||||||
|
if (entry.* & page_size_bit != 0) @panic("paging: descend through a huge-page leaf");
|
||||||
|
return entry.* & address_mask;
|
||||||
|
}
|
||||||
const frame = allocTable();
|
const frame = allocTable();
|
||||||
entry.* = frame | present | writable;
|
entry.* = frame | present | writable;
|
||||||
return frame;
|
return frame;
|
||||||
@@ -96,15 +121,39 @@ fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
|||||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map one 2 MiB huge page `virtual` -> `physical` with `flags` — a leaf at the PD
|
||||||
|
/// level (PS bit set), with no PT beneath it. Both addresses must be 2 MiB-aligned.
|
||||||
|
/// One of these replaces 512 `mapPage`s (and the PT frame they'd need).
|
||||||
|
fn mapHugePage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||||
|
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||||
|
@panic("paging: new higher-half PML4 entry after init");
|
||||||
|
const pdpt = descend(pml4e);
|
||||||
|
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = descend(pdpte);
|
||||||
|
tableAt(pd)[(virtual >> 21) & 0x1FF] = (physical & address_mask) | flags | present | page_size_bit;
|
||||||
|
}
|
||||||
|
|
||||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||||
/// window onto physical memory once the low identity map goes away.
|
/// window onto physical memory once the low identity map goes away. The 2 MiB-
|
||||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
/// aligned interior is mapped with huge pages; the unaligned head/tail with 4 KiB.
|
||||||
|
/// `write_combining` selects the WC memory type (setupPat's PAT entry 4) via the
|
||||||
|
/// PAT bit — bit 7 in a 4 KiB PTE, bit 12 in a huge leaf — for the framebuffer.
|
||||||
|
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64, write_combining: bool) void {
|
||||||
|
const pte_flags = if (write_combining) flags | pte_pat else flags;
|
||||||
|
const huge_flags = if (write_combining) flags | huge_pat else flags;
|
||||||
var address = physical_base & ~@as(u64, page_size - 1);
|
var address = physical_base & ~@as(u64, page_size - 1);
|
||||||
const end = physical_base + len;
|
const end = physical_base + len;
|
||||||
while (address < end) : (address += page_size) {
|
// Head: 4 KiB pages up to the next 2 MiB boundary.
|
||||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
while (address < end and address & (huge_page_size - 1) != 0) : (address += page_size)
|
||||||
}
|
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||||
|
// Interior: 2 MiB huge pages while a whole one still fits.
|
||||||
|
while (address + huge_page_size <= end) : (address += huge_page_size)
|
||||||
|
mapHugePage(pml4, boot_handoff.physicalToVirtual(address), address, huge_flags);
|
||||||
|
// Tail: 4 KiB pages for whatever is left.
|
||||||
|
while (address < end) : (address += page_size)
|
||||||
|
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||||
@@ -118,11 +167,26 @@ fn enableNx() void {
|
|||||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Program this core's PAT so entry 4 (selected by the PAT bit with PCD=PWT=0) is
|
||||||
|
/// **write-combining**, leaving the other seven at their reset types. Nothing else
|
||||||
|
/// in danos sets the PAT bit, so this changes no existing mapping — it only gives
|
||||||
|
/// the framebuffer a write-combining type, which turns its full-screen clear from
|
||||||
|
/// glacial (uncached writes to a GPU BAR, the real-hardware default via MTRRs) into
|
||||||
|
/// a batched burst. Must run on **every** core (PAT is per-logical-processor) — the
|
||||||
|
/// framebuffer mapping lives in the shared kernel half, so a core with the reset
|
||||||
|
/// PAT would see it as write-back and alias. Called from `init` (BSP) and each AP.
|
||||||
|
pub fn setupPat() void {
|
||||||
|
// Reset PAT is PA0=WB PA1=WT PA2=UC- PA3=UC PA4=WB PA5=WT PA6=UC- PA7=UC; flip
|
||||||
|
// PA4 from WB (0x06) to WC (0x01). Type codes: UC=0 WC=1 WT=4 WP=5 WB=6 UC-=7.
|
||||||
|
io.wrmsr(ia32_pat, 0x0007_0401_0007_0406);
|
||||||
|
}
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
/// Build the address space and switch onto it.
|
||||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||||
alloc_frame = allocFrame;
|
alloc_frame = allocFrame;
|
||||||
free_frame = freeFrame;
|
free_frame = freeFrame;
|
||||||
enableNx();
|
enableNx();
|
||||||
|
setupPat(); // BSP: PAT entry 4 = write-combining, for the framebuffer window
|
||||||
const pml4 = allocTable();
|
const pml4 = allocTable();
|
||||||
|
|
||||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||||
@@ -130,13 +194,15 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
|||||||
// mapped on demand (mapMmio) or explicitly below.
|
// mapped on demand (mapMmio) or explicitly below.
|
||||||
for (regions(boot_information.memory_map)) |r| {
|
for (regions(boot_information.memory_map)) |r| {
|
||||||
if (r.kind == .mmio) continue;
|
if (r.kind == .mmio) continue;
|
||||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||||
// the kernel touches directly), RW + NX.
|
// the kernel touches directly), RW + NX. The framebuffer is **write-
|
||||||
|
// combining** (see setupPat) so the console's full-screen clear is a burst,
|
||||||
|
// not millions of uncached single-word writes.
|
||||||
const fb = boot_information.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute, true);
|
||||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||||
|
|
||||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||||
@@ -254,8 +320,12 @@ pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool,
|
|||||||
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
||||||
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
||||||
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
||||||
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64, write_combining: bool) void {
|
||||||
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
// Registers are strong-uncacheable (PCD|PWT). A framebuffer instead wants
|
||||||
|
// write-combining — the PAT bit (bit 7 in a 4 KiB PTE) with PCD=PWT=0 selects PAT
|
||||||
|
// entry 4, which `setupPat` programs to WC — so pixel writes batch into bursts.
|
||||||
|
const cache: u64 = if (write_combining) pte_pat else (pcd | pwt);
|
||||||
|
const flags: u64 = present | user | writable | no_execute | device_grant | cache;
|
||||||
const first = physical & ~@as(u64, page_size - 1);
|
const first = physical & ~@as(u64, page_size - 1);
|
||||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||||
var off: u64 = 0;
|
var off: u64 = 0;
|
||||||
@@ -296,6 +366,57 @@ pub fn mapUserDmaInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The raw leaf entry mapping `virtual` in the address space rooted at `pml4`, or null
|
||||||
|
/// if any level of the walk is absent. **Read-only** — never allocates or descends into
|
||||||
|
/// a missing table (unlike the `map*` paths' `descendUser`). Stops at the first huge
|
||||||
|
/// leaf. For tests and introspection that need a page's actual flag bits.
|
||||||
|
pub fn leafEntryOf(pml4: u64, virtual: u64) ?u64 {
|
||||||
|
const l4 = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (l4 & present == 0) return null;
|
||||||
|
const l3 = tableAt(l4 & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
|
if (l3 & present == 0) return null;
|
||||||
|
if (l3 & page_size_bit != 0) return l3; // 1 GiB leaf
|
||||||
|
const l2 = tableAt(l3 & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
|
if (l2 & present == 0) return null;
|
||||||
|
if (l2 & page_size_bit != 0) return l2; // 2 MiB leaf
|
||||||
|
const l1 = tableAt(l2 & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
|
if (l1 & present == 0) return null;
|
||||||
|
return l1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is the 4 KiB leaf mapping `virtual` write-combining — the PAT bit set with PCD and
|
||||||
|
/// PWT clear, which `setupPat` makes PAT entry 4 (WC)? Null if unmapped. The device
|
||||||
|
/// mapping path (`mapUserDeviceInto`) always uses 4 KiB leaves, so bit 7 (`pte_pat`)
|
||||||
|
/// is the PAT selector in play.
|
||||||
|
pub fn leafIsWriteCombining(pml4: u64, virtual: u64) ?bool {
|
||||||
|
const e = leafEntryOf(pml4, virtual) orelse return null;
|
||||||
|
return (e & pte_pat != 0) and (e & pcd == 0) and (e & pwt == 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map `[physical, physical+len)` into the user half rooted at `pml4` as **shared cacheable
|
||||||
|
/// RAM**: write-back cacheable (RW + NX) for CPU compositing, and carrying `device_grant`
|
||||||
|
/// so teardown (`freeSubtree`) does **not** return the frames to the allocator. The frames
|
||||||
|
/// are owned by a refcounted shared-memory object (system/kernel/ipc-synchronous.zig) and
|
||||||
|
/// freed only when its last capability drops — not when one sharer's address space dies, or
|
||||||
|
/// the other sharers would be left mapping freed RAM. The caller aligns `virtual`/`physical`.
|
||||||
|
pub fn mapUserSharedInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||||
|
const flags: u64 = present | user | writable | no_execute | device_grant; // WB cacheable
|
||||||
|
const first = physical & ~@as(u64, page_size - 1);
|
||||||
|
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||||
|
var off: u64 = 0;
|
||||||
|
while (first + off <= last) : (off += page_size) {
|
||||||
|
const v = virtual + off;
|
||||||
|
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||||
|
const pdpt = descendUser(pml4e);
|
||||||
|
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||||
|
const pd = descendUser(pdpte);
|
||||||
|
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||||
|
const pt = descendUser(pde);
|
||||||
|
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||||
|
invalidate(v);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Create a new address space: a fresh PML4 with an empty user half and the
|
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||||
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||||
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||||
@@ -338,8 +459,8 @@ fn freeSubtree(physical: u64, level: u32) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
/// clear. Walks the 4-level tables, stopping at a 2 MiB huge-page leaf (the physmap
|
||||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
/// uses them). Returns false if unmapped. Used for W^X checks in tests.
|
||||||
pub fn isExecutable(virtual: u64) bool {
|
pub fn isExecutable(virtual: u64) bool {
|
||||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return false;
|
if (pml4e & present == 0) return false;
|
||||||
@@ -347,6 +468,7 @@ pub fn isExecutable(virtual: u64) bool {
|
|||||||
if (pdpte & present == 0) return false;
|
if (pdpte & present == 0) return false;
|
||||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return false;
|
if (pde & present == 0) return false;
|
||||||
|
if (pde & page_size_bit != 0) return pde & no_execute == 0; // 2 MiB huge leaf
|
||||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return false;
|
if (pte & present == 0) return false;
|
||||||
return pte & no_execute == 0;
|
return pte & no_execute == 0;
|
||||||
@@ -385,9 +507,9 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
|||||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||||
/// needs the frame behind a user vaddr to free it).
|
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
||||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return null;
|
if (pml4e & present == 0) return null;
|
||||||
@@ -395,6 +517,8 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
|||||||
if (pdpte & present == 0) return null;
|
if (pdpte & present == 0) return null;
|
||||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
if (pde & present == 0) return null;
|
if (pde & present == 0) return null;
|
||||||
|
if (pde & page_size_bit != 0) // 2 MiB huge leaf: frame base is bits 51:21
|
||||||
|
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
if (pte & present == 0) return null;
|
if (pte & present == 0) return null;
|
||||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
|
|||||||
@@ -17,6 +17,12 @@ const Access = enum { port, mmio };
|
|||||||
var access: Access = .port;
|
var access: Access = .port;
|
||||||
var base: u64 = 0x3F8; // COM1
|
var base: u64 = 0x3F8; // COM1
|
||||||
|
|
||||||
|
/// Whether `init`/`reconfigure` found a *working* UART at `base`. False on a
|
||||||
|
/// legacy-free machine whose COM1 is decoded but dead: writing to it is then a
|
||||||
|
/// no-op, so `write` never spins waiting for a transmit register that will never
|
||||||
|
/// drain. Cleared until proven by the loopback probe.
|
||||||
|
var uart_present: bool = false;
|
||||||
|
|
||||||
fn portOut(p: u16, value: u8) void {
|
fn portOut(p: u16, value: u8) void {
|
||||||
asm volatile ("outb %[value], %[p]"
|
asm volatile ("outb %[value], %[p]"
|
||||||
:
|
:
|
||||||
@@ -57,6 +63,34 @@ pub fn init() void {
|
|||||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||||
setRegister(4, 0x0B); // RTS/DSR set
|
setRegister(4, 0x0B); // RTS/DSR set
|
||||||
|
uart_present = probe();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect a *working* UART by internal loopback: route the transmitter back to
|
||||||
|
/// the receiver (MCR bit 4), send a byte, and check it comes back. A port that is
|
||||||
|
/// merely decoded but has nothing behind it (the common case on a legacy-free
|
||||||
|
/// board that still answers I/O at 0x3F8) never echoes, so this returns false.
|
||||||
|
///
|
||||||
|
/// This matters for speed, not just correctness: a dead UART's line-status
|
||||||
|
/// register reads back 0x00, so its transmit-holding-empty bit never sets, and
|
||||||
|
/// `writeByte` would otherwise spin its full guard — tens of milliseconds — on
|
||||||
|
/// *every* logged byte. On real hardware that alone can add ~a minute to boot.
|
||||||
|
fn probe() bool {
|
||||||
|
const saved_mcr = register(4);
|
||||||
|
setRegister(4, 0x1E); // MCR: LOOP | OUT2 | OUT1 | RTS — internal loopback
|
||||||
|
setRegister(0, 0xAE); // push a distinctive byte into the loopback path
|
||||||
|
var guard: u32 = 0;
|
||||||
|
while (register(5) & 0x01 == 0 and guard < 10_000) : (guard += 1) {} // await Data Ready
|
||||||
|
const echo = register(0);
|
||||||
|
setRegister(4, saved_mcr); // restore the modem-control lines
|
||||||
|
return echo == 0xAE;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a working UART was detected (see `probe`). The log sink stays
|
||||||
|
/// registered regardless — it simply does nothing until this is true — so a UART
|
||||||
|
/// that only `reconfigure` discovers (via SPCR) still starts logging.
|
||||||
|
pub fn present() bool {
|
||||||
|
return uart_present;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||||
@@ -70,15 +104,19 @@ pub fn reconfigure(is_mmio: bool, address: u64) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn writeByte(c: u8) void {
|
fn writeByte(c: u8) void {
|
||||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
// Wait for the transmit-holding register to empty. `write` only reaches here
|
||||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
// for a UART the loopback probe proved live, so this bounds a momentary stall
|
||||||
|
// (e.g. deasserted flow control), not an absent port: ~5000 legacy-port reads
|
||||||
|
// is a few ms — comfortably longer than one 38400-baud byte-time (~260 µs).
|
||||||
var guard: u32 = 0;
|
var guard: u32 = 0;
|
||||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
while (register(5) & 0x20 == 0 and guard < 5_000) : (guard += 1) {}
|
||||||
setRegister(0, c);
|
setRegister(0, c);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
/// Write bytes, translating LF to CRLF so terminals and logs line up. A no-op
|
||||||
|
/// when no working UART was detected, so a dead COM1 costs nothing per byte.
|
||||||
pub fn write(bytes: []const u8) void {
|
pub fn write(bytes: []const u8) void {
|
||||||
|
if (!uart_present) return;
|
||||||
for (bytes) |c| {
|
for (bytes) |c| {
|
||||||
if (c == '\n') writeByte('\r');
|
if (c == '\n') writeByte('\r');
|
||||||
writeByte(c);
|
writeByte(c);
|
||||||
|
|||||||
@@ -148,7 +148,14 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3:
|
|||||||
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
||||||
const deadline = apic.millis() + 100;
|
const deadline = apic.millis() + 100;
|
||||||
while (apic.millis() < deadline) {
|
while (apic.millis() < deadline) {
|
||||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
|
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) {
|
||||||
|
// Cross-check this core's TSC against the BSP's before it joins the run
|
||||||
|
// loop: an unsynchronized TSC must be caught before any task can migrate
|
||||||
|
// onto this core and observe time going backward. No-op unless the TSC is
|
||||||
|
// the clocksource (apic.checkWarpSource).
|
||||||
|
apic.checkWarpSource();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
asm volatile ("pause");
|
asm volatile ("pause");
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
@@ -166,6 +173,7 @@ fn delayMicros(us: u64) void {
|
|||||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||||
const cpu = boot_index;
|
const cpu = boot_index;
|
||||||
|
paging.setupPat(); // this core's PAT: entry 4 = write-combining, to match the BSP
|
||||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||||
idt.loadOnThisCpu(); // the shared IDT
|
idt.loadOnThisCpu(); // the shared IDT
|
||||||
@@ -177,6 +185,11 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
|
|||||||
|
|
||||||
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||||
|
|
||||||
|
// Rendezvous with the BSP for the TSC warp check (no-op unless the TSC is the
|
||||||
|
// clocksource) before joining the run loop, so this core's clock is vetted before
|
||||||
|
// it can run any task.
|
||||||
|
apic.checkWarpTarget();
|
||||||
|
|
||||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,12 +2,15 @@
|
|||||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||||
//! — just pixels.
|
//! — just pixels.
|
||||||
//!
|
//!
|
||||||
//! This is a **bootstrap** console — a stop-gap so early boot has something on
|
//! This is a **bootstrap / fatal-fallback** console. The driver machinery now exists — the
|
||||||
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
|
//! user-space **display service** ([../services/display](../services/display/display.zig),
|
||||||
//! terminal; once the driver machinery exists it becomes a proper graphics device
|
//! docs/display.md) owns the framebuffer in normal operation — so this no longer paints
|
||||||
//! driver and this text-grid crutch goes away. It is therefore kept **separate
|
//! routine status. It exists for the two cases the display service can't cover: **early
|
||||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
|
//! boot**, before the service has claimed the framebuffer, and **fatal errors** (a kernel
|
||||||
//! while this only paints the handful of user-facing status lines and panics.
|
//! panic or a kernel-mode fault), which force it back on (`setSuppressed`) so a dying
|
||||||
|
//! machine's last words reach the screen even over a live display. It is kept **separate
|
||||||
|
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file and
|
||||||
|
//! carries all routine kernel output; this only paints those fatal cases.
|
||||||
//!
|
//!
|
||||||
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
||||||
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
||||||
@@ -20,6 +23,12 @@ const boot_handoff = @import("boot-handoff");
|
|||||||
var con: Console = undefined;
|
var con: Console = undefined;
|
||||||
var con_present: bool = false;
|
var con_present: bool = false;
|
||||||
|
|
||||||
|
/// Set while a user-space display service owns the framebuffer: `write` falls silent so
|
||||||
|
/// the kernel doesn't paint over the compositor. Driven by the display device's
|
||||||
|
/// claim/release (system/kernel/process.zig). The terminal panic/exception paths clear
|
||||||
|
/// it first (`setSuppressed(false)`) — a dying machine's message wins over any display.
|
||||||
|
var suppressed: bool = false;
|
||||||
|
|
||||||
/// Set up the console over `fb`, or mark it absent if there's no usable
|
/// Set up the console over `fb`, or mark it absent if there's no usable
|
||||||
/// framebuffer. Clears the screen when present.
|
/// framebuffer. Clears the screen when present.
|
||||||
pub fn init(fb: boot_handoff.Framebuffer) void {
|
pub fn init(fb: boot_handoff.Framebuffer) void {
|
||||||
@@ -41,13 +50,20 @@ pub fn present() bool {
|
|||||||
return con_present;
|
return con_present;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present,
|
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present, or
|
||||||
/// so it's always safe to call.
|
/// while a display service owns the screen (`suppressed`), so it's always safe to call.
|
||||||
pub fn write(bytes: []const u8) void {
|
pub fn write(bytes: []const u8) void {
|
||||||
if (!con_present) return;
|
if (!con_present or suppressed) return;
|
||||||
for (bytes) |c| con.putChar(c);
|
for (bytes) |c| con.putChar(c);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Quiesce (or resume) the bootstrap console. Set true when a display service claims the
|
||||||
|
/// framebuffer; set false when that claim is released, or by the panic path to force a
|
||||||
|
/// last message onto a screen a (now-irrelevant) service was holding.
|
||||||
|
pub fn setSuppressed(value: bool) void {
|
||||||
|
suppressed = value;
|
||||||
|
}
|
||||||
|
|
||||||
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
||||||
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
||||||
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
||||||
|
|||||||
@@ -36,6 +36,11 @@ var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined;
|
|||||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||||
var count: usize = 0;
|
var count: usize = 0;
|
||||||
|
|
||||||
|
/// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine
|
||||||
|
/// handed over no framebuffer. Lets the process layer recognise the display claim
|
||||||
|
/// (to quiesce the bootstrap console) without threading the id through every caller.
|
||||||
|
var display_device: ?u64 = null;
|
||||||
|
|
||||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||||
@@ -45,10 +50,55 @@ pub var dropped: usize = 0;
|
|||||||
pub fn init(device_tree: *const platform.DeviceTree) void {
|
pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||||
count = 0;
|
count = 0;
|
||||||
dropped = 0;
|
dropped = 0;
|
||||||
|
display_device = null;
|
||||||
for (&claimed) |*c| c.* = null;
|
for (&claimed) |*c| c.* = null;
|
||||||
walk(device_tree.root, device_abi.no_parent);
|
walk(device_tree.root, device_abi.no_parent);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Publish the loader's framebuffer as a `display` device — a root-level node with one
|
||||||
|
/// write-combining `memory` resource over the linear framebuffer and its geometry in
|
||||||
|
/// `.display`. The framebuffer is *not* firmware-discovered (it rides the
|
||||||
|
/// [[boot-handoff]], not the device tree), so it is seeded explicitly, after `init`.
|
||||||
|
/// Returns the new device id, or null when there is no framebuffer (headless) or the
|
||||||
|
/// table is full. Idempotent-ish: only ever call once per boot.
|
||||||
|
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32) ?u64 {
|
||||||
|
if (base == 0 or width == 0 or height == 0) return null; // headless
|
||||||
|
if (count >= maximum_devices) {
|
||||||
|
dropped += 1;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||||
|
d.id = count;
|
||||||
|
d.parent = device_abi.no_parent;
|
||||||
|
d.class = @intFromEnum(device_abi.DeviceClass.display);
|
||||||
|
d.pci_class = device_abi.no_pci_class;
|
||||||
|
d.resource_count = 1;
|
||||||
|
d.resources[0] = .{
|
||||||
|
.kind = @intFromEnum(device_abi.ResourceKind.memory),
|
||||||
|
.start = base,
|
||||||
|
.len = @as(u64, height) * pitch,
|
||||||
|
.flags = device_abi.resource_flag_write_combining,
|
||||||
|
};
|
||||||
|
d.display = .{ .width = width, .height = height, .pitch = pitch, .format = format };
|
||||||
|
devices[count] = d;
|
||||||
|
display_device = d.id;
|
||||||
|
count += 1;
|
||||||
|
return d.id;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The id of the seeded framebuffer device, or null when none was seeded.
|
||||||
|
pub fn displayDevice() ?u64 {
|
||||||
|
return display_device;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the framebuffer device is currently claimed by some process. The bootstrap
|
||||||
|
/// console uses this (via the process layer) to fall silent while a display service
|
||||||
|
/// owns the screen, and to resume if that service dies and its claim is released.
|
||||||
|
pub fn displayClaimed() bool {
|
||||||
|
const id = display_device orelse return false;
|
||||||
|
return ownerOf(id) != null;
|
||||||
|
}
|
||||||
|
|
||||||
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||||
/// assigned it down to its children as their parent.
|
/// assigned it down to its children as their parent.
|
||||||
fn walk(node: *platform.Device, parent_id: u64) void {
|
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||||
@@ -189,7 +239,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
|||||||
if (!ok) return error.NotContained;
|
if (!ok) return error.NotContained;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Idempotent on exact match (docs/m19-m20-plan.md decision 3): a restarted
|
// Idempotent on exact match (docs/device-manager.md): a restarted
|
||||||
// registering bus re-registers what it rediscovers, and the table has no
|
// registering bus re-registers what it rediscovers, and the table has no
|
||||||
// unregister — an identical (class, identity, resources) child under the
|
// unregister — an identical (class, identity, resources) child under the
|
||||||
// same parent returns the existing id instead of appending a duplicate.
|
// same parent returns the existing id instead of appending a duplicate.
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ const architecture = @import("architecture");
|
|||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
|
const pmm = @import("pmm.zig");
|
||||||
|
|
||||||
const page_size = abi.page_size;
|
const page_size = abi.page_size;
|
||||||
const Task = scheduler.Task;
|
const Task = scheduler.Task;
|
||||||
@@ -37,7 +38,9 @@ const Task = scheduler.Task;
|
|||||||
pub const MESSAGE_MAXIMUM: usize = 256;
|
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||||
|
|
||||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||||
pub const maximum_services = 8;
|
// The name registry is indexed directly by ServiceId, so this must exceed the
|
||||||
|
// largest id (currently fat = 8). Sized with headroom for new services.
|
||||||
|
pub const maximum_services = 16;
|
||||||
|
|
||||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||||
pub const EBADF: i64 = 1; // bad handle
|
pub const EBADF: i64 = 1; // bad handle
|
||||||
@@ -87,6 +90,10 @@ const user_half_end: u64 = 0x0000_8000_0000_0000;
|
|||||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||||
pub const Endpoint = struct {
|
pub const Endpoint = struct {
|
||||||
refcount: u32 = 1,
|
refcount: u32 = 1,
|
||||||
|
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||||
|
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||||
|
owner: u32 = 0,
|
||||||
|
dead: bool = false,
|
||||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||||
sender_head: ?*Task = null,
|
sender_head: ?*Task = null,
|
||||||
@@ -107,10 +114,30 @@ pub const Endpoint = struct {
|
|||||||
|
|
||||||
pub fn createIpcEndpoint() ?*Endpoint {
|
pub fn createIpcEndpoint() ?*Endpoint {
|
||||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||||
endpoint.* = .{};
|
endpoint.* = .{ .owner = scheduler.currentId() };
|
||||||
return endpoint;
|
return endpoint;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
||||||
|
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
||||||
|
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
||||||
|
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
||||||
|
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
||||||
|
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||||
|
for (®istry) |*slot| {
|
||||||
|
const endpoint = slot.* orelse continue;
|
||||||
|
if (endpoint.owner != task_id) continue;
|
||||||
|
endpoint.dead = true;
|
||||||
|
while (dequeueSender(endpoint)) |sender| {
|
||||||
|
sender.ipc_status = -EPEER;
|
||||||
|
sender.ipc_received_cap = abi.no_cap;
|
||||||
|
scheduler.readyLocked(sender);
|
||||||
|
}
|
||||||
|
slot.* = null;
|
||||||
|
dropRef(endpoint);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||||
pub fn dropRef(endpoint: *Endpoint) void {
|
pub fn dropRef(endpoint: *Endpoint) void {
|
||||||
@@ -121,6 +148,43 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- capability objects: what a handle-table entry can name ------------------
|
||||||
|
|
||||||
|
/// The `kind` tag on a `scheduler.HandleObject` — which capability object a handle names.
|
||||||
|
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||||
|
pub const handle_kind_endpoint: u8 = 0;
|
||||||
|
pub const handle_kind_shm: u8 = 1;
|
||||||
|
|
||||||
|
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||||
|
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||||
|
/// contiguous physical base, `pages` its length. A sharer's address-space teardown never
|
||||||
|
/// reclaims these frames (the mapping carries `device_grant`); this object owns them.
|
||||||
|
pub const ShmObject = struct {
|
||||||
|
refcount: u32 = 1,
|
||||||
|
phys: u64,
|
||||||
|
pages: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Wrap `pages` contiguous frames at `phys` (already allocated + zeroed by the caller) in a
|
||||||
|
/// refcounted shm object, or null if the heap is out of room.
|
||||||
|
pub fn createShm(phys: u64, pages: usize) ?*ShmObject {
|
||||||
|
const shm = heap.allocator().create(ShmObject) catch return null;
|
||||||
|
shm.* = .{ .phys = phys, .pages = pages };
|
||||||
|
return shm;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop a shared-memory reference; when the last one goes, return its frames to the
|
||||||
|
/// allocator and free the object. (The mappings themselves are torn down with each
|
||||||
|
/// sharer's address space; `device_grant` keeps that from freeing the frames early.)
|
||||||
|
pub fn dropShmRef(shm: *ShmObject) void {
|
||||||
|
if (shm.refcount > 1) {
|
||||||
|
shm.refcount -= 1;
|
||||||
|
} else {
|
||||||
|
for (0..shm.pages) |i| pmm.free(shm.phys + i * page_size);
|
||||||
|
heap.allocator().destroy(shm);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||||
|
|
||||||
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||||
@@ -220,11 +284,25 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
|||||||
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
|
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
|
||||||
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
|
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
|
||||||
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||||
const endpoint = resolveHandle(from, cap) orelse return -EBADF;
|
if (cap >= from.handles.len) return -EBADF;
|
||||||
endpoint.refcount += 1;
|
const entry = from.handles[@intCast(cap)] orelse return -EBADF;
|
||||||
const handle = installHandle(to, endpoint);
|
// Bump the named object's refcount (a copy, not a move — the sender keeps its handle),
|
||||||
|
// dispatching by kind so both endpoints and shared-memory regions can travel with a
|
||||||
|
// message.
|
||||||
|
switch (entry.kind) {
|
||||||
|
handle_kind_endpoint => {
|
||||||
|
const e: *Endpoint = @ptrCast(@alignCast(entry.ptr));
|
||||||
|
e.refcount += 1;
|
||||||
|
},
|
||||||
|
handle_kind_shm => {
|
||||||
|
const s: *ShmObject = @ptrCast(@alignCast(entry.ptr));
|
||||||
|
s.refcount += 1;
|
||||||
|
},
|
||||||
|
else => return -EBADF,
|
||||||
|
}
|
||||||
|
const handle = installEntry(to, entry);
|
||||||
if (handle < 0) {
|
if (handle < 0) {
|
||||||
dropRef(endpoint); // undo the bump; the receiver had no room
|
dropEntry(entry); // undo the bump; the receiver had no room
|
||||||
return -ENOSPC;
|
return -ENOSPC;
|
||||||
}
|
}
|
||||||
return handle;
|
return handle;
|
||||||
@@ -239,6 +317,7 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
|
|||||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
if (endpoint.dead) return -EPEER; // the service that owned this endpoint is gone — don't block
|
||||||
|
|
||||||
const me = scheduler.current();
|
const me = scheduler.current();
|
||||||
me.ipc_send_ptr = message_ptr;
|
me.ipc_send_ptr = message_ptr;
|
||||||
@@ -414,36 +493,68 @@ pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
|||||||
|
|
||||||
// --- per-process handle table + name registry -------------------------------
|
// --- per-process handle table + name registry -------------------------------
|
||||||
|
|
||||||
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
/// Install a capability object (kind + pointer) in task `t`'s handle table; returns the
|
||||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
/// small-int handle or -ENOSPC. The caller has already taken/holds the reference the slot
|
||||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
/// represents.
|
||||||
|
fn installEntry(t: *Task, entry: scheduler.HandleObject) i64 {
|
||||||
for (&t.handles, 0..) |*slot, i| {
|
for (&t.handles, 0..) |*slot, i| {
|
||||||
if (slot.* == null) {
|
if (slot.* == null) {
|
||||||
slot.* = @ptrCast(endpoint);
|
slot.* = entry;
|
||||||
return @intCast(i);
|
return @intCast(i);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return -ENOSPC;
|
return -ENOSPC;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
/// Install an endpoint handle. The common case; keeps the endpoint callers' signature.
|
||||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||||
if (h >= t.handles.len) return null;
|
return installEntry(t, .{ .kind = handle_kind_endpoint, .ptr = @ptrCast(endpoint) });
|
||||||
const slot = t.handles[@intCast(h)] orelse return null;
|
|
||||||
return @ptrCast(@alignCast(slot));
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
/// Install a shared-memory handle.
|
||||||
/// exit path so a dead server's endpoints don't linger referenced.
|
pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 {
|
||||||
|
return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) });
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
|
||||||
|
/// (e.g. an shm handle used where an endpoint is expected).
|
||||||
|
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||||
|
if (h >= t.handles.len) return null;
|
||||||
|
const entry = t.handles[@intCast(h)] orelse return null;
|
||||||
|
if (entry.kind != handle_kind_endpoint) return null;
|
||||||
|
return @ptrCast(@alignCast(entry.ptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve a handle to its shared-memory object, or null if out of range, unused, or not
|
||||||
|
/// an shm handle.
|
||||||
|
pub fn resolveShm(t: *Task, h: u64) ?*ShmObject {
|
||||||
|
if (h >= t.handles.len) return null;
|
||||||
|
const entry = t.handles[@intCast(h)] orelse return null;
|
||||||
|
if (entry.kind != handle_kind_shm) return null;
|
||||||
|
return @ptrCast(@alignCast(entry.ptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop every capability reference an exiting task holds, dispatching by kind so a dead
|
||||||
|
/// task's endpoints *and* shared-memory regions are released correctly. Called from the
|
||||||
|
/// scheduler exit path.
|
||||||
pub fn closeHandles(t: *Task) void {
|
pub fn closeHandles(t: *Task) void {
|
||||||
for (&t.handles) |*slot| {
|
for (&t.handles) |*slot| {
|
||||||
if (slot.*) |p| {
|
if (slot.*) |entry| {
|
||||||
dropRef(@ptrCast(@alignCast(p)));
|
dropEntry(entry);
|
||||||
slot.* = null;
|
slot.* = null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Drop the reference a handle-table entry represents, by kind.
|
||||||
|
fn dropEntry(entry: scheduler.HandleObject) void {
|
||||||
|
switch (entry.kind) {
|
||||||
|
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||||
|
handle_kind_shm => dropShmRef(@ptrCast(@alignCast(entry.ptr))),
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||||
|
|
||||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||||
|
|||||||
+134
-54
@@ -5,9 +5,11 @@ const parameters = @import("parameters");
|
|||||||
const architecture = @import("architecture");
|
const architecture = @import("architecture");
|
||||||
const console = @import("console.zig");
|
const console = @import("console.zig");
|
||||||
const log = @import("log.zig");
|
const log = @import("log.zig");
|
||||||
|
const wall_clock = @import("wall-clock.zig");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
|
const sync = @import("sync.zig");
|
||||||
const process = @import("process.zig");
|
const process = @import("process.zig");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
const irq = @import("irq.zig");
|
const irq = @import("irq.zig");
|
||||||
@@ -59,16 +61,34 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||||
// with port-0x80 checkpoints as the only progress signal.
|
// with port-0x80 checkpoints as the only progress signal.
|
||||||
architecture.serialInit();
|
//
|
||||||
log.addSink(architecture.serialWrite);
|
// Serial is compiled in only under -Dserial (build.zig): a real machine often
|
||||||
|
// has no live legacy COM1, and the log survives in the RAM buffer (below) and
|
||||||
|
// is flushed to disk — so serial is now a QEMU/dev convenience the flashable
|
||||||
|
// image leaves out. When it *is* built in, `serialInit`'s loopback probe still
|
||||||
|
// guards against a dead port (so a -Dserial image is safe on real hardware).
|
||||||
|
if (build_options.serial) {
|
||||||
|
architecture.serialInit();
|
||||||
|
log.addSink(architecture.serialWrite);
|
||||||
|
}
|
||||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||||
|
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||||
|
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||||
|
// see it on a headless/real machine with no host capturing serial.
|
||||||
|
log.addSink(log.ramSink);
|
||||||
|
|
||||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||||
|
//
|
||||||
|
// The console is brought up *after* paging (below), not here: its one-time
|
||||||
|
// full-screen clear then runs on the kernel's **write-combining** mapping of the
|
||||||
|
// framebuffer instead of the loader's uncached one — a fast burst rather than
|
||||||
|
// millions of uncached writes on real hardware. Until then, on-screen output is
|
||||||
|
// absent (an early panic still lands in the serial/RAM log); the trade is worth
|
||||||
|
// a near-instant boot. `console.write` is a safe no-op while the console is down.
|
||||||
const fb = boot_information.framebuffer;
|
const fb = boot_information.framebuffer;
|
||||||
console.init(fb);
|
|
||||||
|
|
||||||
log.checkpoint(cp_entry);
|
log.checkpoint(cp_entry);
|
||||||
|
|
||||||
@@ -77,12 +97,12 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
architecture.setFaultHandler(onException);
|
architecture.setFaultHandler(onException);
|
||||||
architecture.init();
|
architecture.init();
|
||||||
|
|
||||||
status("danos: initialising kernel...\n");
|
status("/system/kernel: initialising kernel...\n");
|
||||||
log.write(if (console.present())
|
if (build_options.serial) log.write(if (architecture.serialPresent())
|
||||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
"/system/kernel: serial console online (COM1)\n"
|
||||||
else
|
else
|
||||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
"/system/kernel: no serial UART (COM1 absent) -> log kept in RAM/debugcon\n");
|
||||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
||||||
@@ -105,7 +125,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
const total_bytes = total_pages * abi.page_size;
|
const total_bytes = total_pages * abi.page_size;
|
||||||
const gib = 1 << 30;
|
const gib = 1 << 30;
|
||||||
|
|
||||||
log.write("\ndanos: physical memory\n");
|
log.write("\n/system/kernel: physical memory\n");
|
||||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||||
@@ -119,7 +139,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
||||||
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
|
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
|
||||||
const s1 = pmm.stats();
|
const s1 = pmm.stats();
|
||||||
log.print("\ndanos: frame allocator online\n", .{});
|
log.print("\n/system/kernel: frame allocator online\n", .{});
|
||||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||||
const f0 = pmm.alloc();
|
const f0 = pmm.alloc();
|
||||||
const f1 = pmm.alloc();
|
const f1 = pmm.alloc();
|
||||||
@@ -133,14 +153,24 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||||
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
||||||
log.checkpoint(cp_paging);
|
log.checkpoint(cp_paging);
|
||||||
log.print("\ndanos: paging enabled\n", .{});
|
log.print("\n/system/kernel: paging enabled\n", .{});
|
||||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||||
|
|
||||||
|
// Now on our own tables, the framebuffer window is write-combining: bring up the
|
||||||
|
// on-screen console and clear it to a blank canvas (a fast burst here, not the loader's
|
||||||
|
// uncached crawl). Routine boot output goes only to the log; this console now exists for
|
||||||
|
// early-boot and fatal (`fatal`/panic) output, until the display service takes over.
|
||||||
|
console.init(fb);
|
||||||
|
log.write(if (console.present())
|
||||||
|
"/system/kernel: framebuffer ready (early-boot + fatal fallback; the display service drives it in normal operation)\n"
|
||||||
|
else
|
||||||
|
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||||
heap.init();
|
heap.init();
|
||||||
log.checkpoint(cp_heap);
|
log.checkpoint(cp_heap);
|
||||||
log.write("\ndanos: kernel heap online\n");
|
log.write("\n/system/kernel: kernel heap online\n");
|
||||||
// Measure the amount of resources the kernel is actually using
|
// Measure the amount of resources the kernel is actually using
|
||||||
const s2 = pmm.stats();
|
const s2 = pmm.stats();
|
||||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||||
@@ -156,7 +186,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
};
|
};
|
||||||
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
||||||
var device_tree = devtree;
|
var device_tree = devtree;
|
||||||
log.write("\ndanos: device discovery online\n");
|
log.write("\n/system/kernel: device discovery online\n");
|
||||||
device_tree.dump(log.write);
|
device_tree.dump(log.write);
|
||||||
|
|
||||||
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||||
@@ -164,28 +194,27 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
devices_broker.init(&device_tree);
|
devices_broker.init(&device_tree);
|
||||||
if (devices_broker.dropped > 0) {
|
if (devices_broker.dropped > 0) {
|
||||||
// Otherwise entirely silent: drivers would just never see that hardware.
|
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
log.print("/system/kernel: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Publish the loader's framebuffer as a claimable `display` device, so a
|
||||||
|
// user-space display service can take it over the same claim + mmio_map path as
|
||||||
|
// any other hardware (it is not firmware-discovered; it rides the boot handoff).
|
||||||
|
if (devices_broker.seedDisplay(fb.base, fb.width, fb.height, fb.pitch, @intFromEnum(fb.format))) |display_id| {
|
||||||
|
log.print("/system/kernel: framebuffer device {d} seeded ({d}x{d}, pitch {d}, write-combining)\n", .{ display_id, fb.width, fb.height, fb.pitch });
|
||||||
}
|
}
|
||||||
|
|
||||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||||
irq.init();
|
irq.init();
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||||
|
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||||
const pw = platform.powerInformation();
|
const pw = platform.powerInformation();
|
||||||
log.write("danos: power\n");
|
log.write("/system/kernel: power\n");
|
||||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||||
if (pw.s5) |s| {
|
|
||||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
|
||||||
} else {
|
|
||||||
log.write(" S5 slp_typ : (not found)\n");
|
|
||||||
}
|
|
||||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||||
|
|
||||||
// AML namespace parse integrity: consumed should equal total.
|
|
||||||
const am = platform.amlStats();
|
|
||||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
|
||||||
|
|
||||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||||
@@ -221,7 +250,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
});
|
});
|
||||||
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
||||||
|
|
||||||
log.write("danos: platform\n");
|
log.write("/system/kernel: platform\n");
|
||||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
||||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
||||||
@@ -237,7 +266,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
if (platform.cpusDropped() > 0)
|
if (platform.cpusDropped() > 0)
|
||||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||||
} else |err| {
|
} else |err| {
|
||||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)});
|
||||||
}
|
}
|
||||||
log.checkpoint(cp_discovery);
|
log.checkpoint(cp_discovery);
|
||||||
|
|
||||||
@@ -248,19 +277,43 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// Register the current context as the first task before enabling preemption.
|
// Register the current context as the first task before enabling preemption.
|
||||||
scheduler.init(4);
|
scheduler.init(4);
|
||||||
log.checkpoint(cp_scheduler);
|
log.checkpoint(cp_scheduler);
|
||||||
log.write("\ndanos: scheduler online\n");
|
log.write("\n/system/kernel: scheduler online\n");
|
||||||
|
|
||||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||||
// the timer preempts among tasks.
|
// the timer preempts among tasks.
|
||||||
architecture.startTimer();
|
architecture.startTimer();
|
||||||
architecture.enableInterrupts();
|
architecture.enableInterrupts();
|
||||||
log.checkpoint(cp_timer);
|
log.checkpoint(cp_timer);
|
||||||
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||||
|
|
||||||
|
// The tsc-sync test forces the TSC clocksource on before the cores come up, so the
|
||||||
|
// TSC + warp-check path is exercised even under TCG (which won't advertise an
|
||||||
|
// invariant TSC). Inert in a normal build (docs/timers.md).
|
||||||
|
if (build_options.test_case) |tc| {
|
||||||
|
if (std.mem.eql(u8, tc, "tsc-sync")) architecture.forceTscClocksourceForTest();
|
||||||
|
}
|
||||||
|
|
||||||
// Wake the other cores (application processors). A no-op on a single-core
|
// Wake the other cores (application processors). A no-op on a single-core
|
||||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). The
|
||||||
|
// per-core TSC warp check rides this: each AP is vetted before it joins the run
|
||||||
|
// loop (docs/timers.md).
|
||||||
bringUpSecondaries();
|
bringUpSecondaries();
|
||||||
|
|
||||||
|
// Report the monotonic clock's final reliability, now the warp check has run on
|
||||||
|
// every core. On real Intel/AMD this is the invariant, synchronized TSC; a bare
|
||||||
|
// VM (no invariant bit) or a machine whose cores' TSCs skew uses the HPET instead.
|
||||||
|
log.print("/system/kernel: clocksource {s} (TSC invariant: {s}, synchronized: {s})\n", .{
|
||||||
|
architecture.clockSourceName(),
|
||||||
|
if (architecture.clockInvariant()) "yes" else "no",
|
||||||
|
if (architecture.clockSynchronized()) "yes" else "no",
|
||||||
|
});
|
||||||
|
if (!architecture.clockSynchronized())
|
||||||
|
log.write("/system/kernel: WARNING: per-core TSCs are not synchronized; monotonic clock moved off the TSC\n");
|
||||||
|
|
||||||
|
// Anchor wall-clock time: read the RTC once, now the monotonic clock is final.
|
||||||
|
wall_clock.init();
|
||||||
|
log.print("/system/kernel: wall clock {d} (Unix epoch seconds, UTC, from the RTC)\n", .{wall_clock.nowSeconds()});
|
||||||
|
|
||||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||||
// Normal builds fall through to the idle halt.
|
// Normal builds fall through to the idle halt.
|
||||||
if (build_options.test_case) |case| {
|
if (build_options.test_case) |case| {
|
||||||
@@ -269,7 +322,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
}
|
}
|
||||||
|
|
||||||
log.checkpoint(cp_running);
|
log.checkpoint(cp_running);
|
||||||
status("kernel initialised.\n");
|
status("/system/kernel: initialised.\n");
|
||||||
|
|
||||||
// Publish the initial-ramdisk so user space can `system_spawn` its bundled
|
// Publish the initial-ramdisk so user space can `system_spawn` its bundled
|
||||||
// binaries by name. The kernel no longer launches them itself: init is the
|
// binaries by name. The kernel no longer launches them itself: init is the
|
||||||
@@ -282,10 +335,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// manager then discovers the hardware and spawns each driver. init runs on its own
|
// manager then discovers the hardware and spawns each driver. init runs on its own
|
||||||
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
||||||
if (boot_information.init_len != 0) {
|
if (boot_information.init_len != 0) {
|
||||||
status("starting /system/services/init...\n");
|
status("/system/kernel: starting /system/services/init...\n");
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||||
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
||||||
statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)});
|
statusPrint("/system/kernel: /system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||||
};
|
};
|
||||||
} else {
|
} else {
|
||||||
status("no /system/services/init on the boot volume.\n");
|
status("no /system/services/init on the boot volume.\n");
|
||||||
@@ -294,7 +347,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
// Become the idle task: drop below every real task and halt until an
|
// Become the idle task: drop below every real task and halt until an
|
||||||
// interrupt. The timer keeps preempting into init and any other work.
|
// interrupt. The timer keeps preempting into init and any other work.
|
||||||
scheduler.setPriority(0);
|
scheduler.setPriority(0);
|
||||||
status("\nkernel idle; user space is running.\n");
|
status("\n/system/kernel: kernel idle; user space is running.\n");
|
||||||
architecture.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -321,7 +374,7 @@ fn bringUpSecondaries() void {
|
|||||||
// vector addresses it). It's kept for the system's life — armed only during a
|
// vector addresses it). It's kept for the system's life — armed only during a
|
||||||
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
|
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
|
||||||
if (ap_trampoline_page == 0) {
|
if (ap_trampoline_page == 0) {
|
||||||
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
log.write("/system/kernel: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
architecture.setTrampolinePage(ap_trampoline_page);
|
architecture.setTrampolinePage(ap_trampoline_page);
|
||||||
@@ -333,7 +386,7 @@ fn bringUpSecondaries() void {
|
|||||||
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
log.print("\n/system/kernel: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||||
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
||||||
for (cores[1..], 1..) |core, index| {
|
for (cores[1..], 1..) |core, index| {
|
||||||
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
||||||
@@ -344,7 +397,7 @@ fn bringUpSecondaries() void {
|
|||||||
// This core's dedicated fault stack — allocated only now that the core is
|
// This core's dedicated fault stack — allocated only now that the core is
|
||||||
// real, rather than reserved statically for every possible core.
|
// real, rather than reserved statically for every possible core.
|
||||||
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
||||||
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
log.print("/system/kernel: cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||||
@@ -353,21 +406,32 @@ fn bringUpSecondaries() void {
|
|||||||
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
||||||
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||||
pc.online = true;
|
pc.online = true;
|
||||||
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
log.print("/system/kernel: cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (attempt == maximum_wake_attempts)
|
if (attempt == maximum_wake_attempts)
|
||||||
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
log.print("/system/kernel: cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
/// A user-facing status line. Now that the user-space **display service** owns the
|
||||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
/// framebuffer in normal operation (docs/display.md), routine kernel output goes to the
|
||||||
/// touches the framebuffer.
|
/// diagnostic `log` (serial/debugcon/RAM) *only* — never to the on-screen console, which
|
||||||
|
/// the compositor is about to paint over. For a message that must reach the screen even so
|
||||||
|
/// — a panic or a fatal fault, when the machine is going down — use `fatal`.
|
||||||
fn status(message: []const u8) void {
|
fn status(message: []const u8) void {
|
||||||
log.write(message);
|
log.write(message);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A fatal, user-facing message: to the diagnostic log *and* the on-screen console, forcing
|
||||||
|
/// the console back on (`setSuppressed(false)`) first — a dying machine's last words outrank
|
||||||
|
/// any display service holding the framebuffer. The console is otherwise silent in normal
|
||||||
|
/// operation (see `status`); it exists now only for early-boot and fatal output.
|
||||||
|
fn fatal(message: []const u8) void {
|
||||||
|
log.write(message);
|
||||||
|
console.setSuppressed(false);
|
||||||
console.write(message);
|
console.write(message);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -376,6 +440,11 @@ fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
|||||||
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn fatalPrint(comptime fmt: []const u8, args: anytype) void {
|
||||||
|
var buffer: [256]u8 = undefined;
|
||||||
|
fatal(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
/// Frames (4 KiB pages) to whole MiB.
|
/// Frames (4 KiB pages) to whole MiB.
|
||||||
fn mib(pages: u64) u64 {
|
fn mib(pages: u64) u64 {
|
||||||
return pages * abi.page_size / (1024 * 1024);
|
return pages * abi.page_size / (1024 * 1024);
|
||||||
@@ -427,7 +496,7 @@ fn exitReasonForVector(vector: u64) abi.ExitReason {
|
|||||||
|
|
||||||
fn onException(state: *const architecture.CpuState) noreturn {
|
fn onException(state: *const architecture.CpuState) noreturn {
|
||||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||||
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
statusPrint("\n/system/kernel: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||||
@@ -436,16 +505,25 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
|||||||
|
|
||||||
log.checkpoint(cp_exception);
|
log.checkpoint(cp_exception);
|
||||||
const core = scheduler.currentCpuIndex();
|
const core = scheduler.currentCpuIndex();
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
// The machine is going down: paint the exception on screen too — `fatalPrint` forces the
|
||||||
// top of the diagnostic log.
|
// console back on even if a display service was holding the framebuffer — on top of the
|
||||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
// diagnostic log.
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
// Name the culprit: which task, and whether it faulted in ring 3 (a process the
|
||||||
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
// kernel would normally kill — landing here means it had no address space) or ring 0
|
||||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
// (the trusted base itself). Without this the fatal report is anonymous.
|
||||||
|
fatalPrint(" task : {d} ({s}), {s}\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe(), if (architecture.fromUser(state)) "ring 3 (user)" else "ring 0 (kernel)" });
|
||||||
|
fatalPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
|
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||||
|
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||||
|
if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||||
|
|
||||||
var buffer: [128]u8 = undefined;
|
var buffer: [128]u8 = undefined;
|
||||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||||
|
// Free the BKL if this core held it (a kernel-mode fault, or a nested fault in the
|
||||||
|
// recovery teardown), so halting this one core doesn't deadlock every other core on
|
||||||
|
// the lock. Only that core stops; the rest — and the supervisor — keep running.
|
||||||
|
sync.releaseIfHeldHere();
|
||||||
architecture.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -457,9 +535,11 @@ pub const panic = std.debug.FullPanic(struct {
|
|||||||
_ = first_trace_address;
|
_ = first_trace_address;
|
||||||
log.checkpoint(cp_panic);
|
log.checkpoint(cp_panic);
|
||||||
log.recordPanic(message);
|
log.recordPanic(message);
|
||||||
status("\nKERNEL PANIC: ");
|
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||||
status(message);
|
fatal(message);
|
||||||
status("\n");
|
fatal("\n");
|
||||||
|
fatalPrint(" task : {d} ({s})\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe() });
|
||||||
|
sync.releaseIfHeldHere(); // don't deadlock the other cores on the lock we may hold
|
||||||
architecture.halt();
|
architecture.halt();
|
||||||
}
|
}
|
||||||
}.panic);
|
}.panic);
|
||||||
|
|||||||
@@ -41,6 +41,39 @@ pub fn write(bytes: []const u8) void {
|
|||||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- the RAM sink: a retained copy of the whole diagnostic stream ------------
|
||||||
|
//
|
||||||
|
// A fixed in-image buffer that accumulates every logged byte, so a user program
|
||||||
|
// (`log-flush`, and init at shutdown) can read it back through `klog_read` and
|
||||||
|
// persist it to a file — the boot log survives on a headless/real machine that
|
||||||
|
// has no host capturing serial. It is a *sink like any other*: register it with
|
||||||
|
// `addSink(ramSink)` at boot. No allocation (works pre-heap and in a panic).
|
||||||
|
//
|
||||||
|
// It fills linearly and stops when full: the earliest output — the most valuable
|
||||||
|
// for diagnosing a boot — is kept, and the tail is still on the live serial sink.
|
||||||
|
// 256 KiB comfortably holds a full boot plus a long run (a boot is ~15 KiB).
|
||||||
|
|
||||||
|
const ram_capacity = 256 * 1024;
|
||||||
|
var ram_buffer: [ram_capacity]u8 = undefined;
|
||||||
|
var ram_len: usize = 0;
|
||||||
|
|
||||||
|
/// The RAM sink. Best-effort and self-guarding like every sink: appends what fits
|
||||||
|
/// and silently drops the rest once full. (Concurrency matches the other sinks —
|
||||||
|
/// the dominant writer, debug_write, already holds the kernel lock; a rare torn
|
||||||
|
/// append on a kernel-internal line is an accepted diagnostic imperfection.)
|
||||||
|
pub fn ramSink(bytes: []const u8) void {
|
||||||
|
const n = @min(ram_buffer.len - ram_len, bytes.len);
|
||||||
|
if (n != 0) {
|
||||||
|
@memcpy(ram_buffer[ram_len..][0..n], bytes[0..n]);
|
||||||
|
ram_len += n;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The accumulated log so far — what `klog_read` copies out.
|
||||||
|
pub fn ramSnapshot() []const u8 {
|
||||||
|
return ram_buffer[0..ram_len];
|
||||||
|
}
|
||||||
|
|
||||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||||
/// this is safe to call from interrupt context and from a panic.
|
/// this is safe to call from interrupt context and from a panic.
|
||||||
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||||
|
|||||||
+172
-19
@@ -28,12 +28,14 @@ const parameters = @import("parameters");
|
|||||||
const architecture = @import("architecture");
|
const architecture = @import("architecture");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
|
const console = @import("console.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const ipc = @import("ipc-synchronous.zig");
|
const ipc = @import("ipc-synchronous.zig");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
const irq = @import("irq.zig");
|
const irq = @import("irq.zig");
|
||||||
const initial_ramdisk = @import("initial-ramdisk");
|
const initial_ramdisk = @import("initial-ramdisk");
|
||||||
const log = @import("log.zig");
|
const log = @import("log.zig");
|
||||||
|
const wall_clock = @import("wall-clock.zig");
|
||||||
|
|
||||||
const page_size = abi.page_size;
|
const page_size = abi.page_size;
|
||||||
const SystemCall = abi.SystemCall;
|
const SystemCall = abi.SystemCall;
|
||||||
@@ -79,9 +81,23 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
|||||||
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
|
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
|
||||||
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
|
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
|
||||||
|
|
||||||
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
/// The shared-memory arena: where `shm_create`/`shm_map` place shared cacheable regions, in
|
||||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
/// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by
|
||||||
const maximum_mmap_pages = 256;
|
/// a refcounted shm object and freed when its last capability drops, not on teardown, so the
|
||||||
|
/// mapping carries `device_grant`. Per-process cursor in `Task.shm_map_next` (docs/display-v2.md).
|
||||||
|
pub const shm_arena_base: u64 = 0x0000_7300_0000_0000;
|
||||||
|
pub const shm_arena_end: u64 = shm_arena_base + (256 << 20); // 256 MiB per process
|
||||||
|
|
||||||
|
/// Largest single `shm_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
|
||||||
|
/// also an overflow guard on the page count. shm frames are contiguous (like DMA), so this
|
||||||
|
/// bounds the contiguous allocation asked of the frame allocator.
|
||||||
|
const maximum_shm_pages = 8192;
|
||||||
|
|
||||||
|
/// Largest single `mmap` grant, in pages (32 MiB). Big enough for a display service's
|
||||||
|
/// back buffer at up to 4K (3840x2160x4 ≈ 8100 pages); the user heap otherwise grows in
|
||||||
|
/// small chunks. `systemMmap` maps page by page with rollback, so this is only a sanity
|
||||||
|
/// bound (and an overflow guard on the page count), not the size of any scratch array.
|
||||||
|
const maximum_mmap_pages = 8192;
|
||||||
|
|
||||||
/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn
|
/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn
|
||||||
/// parameters ("you are the driver for device 12"), not bulk data — IPC carries
|
/// parameters ("you are the driver for device 12"), not bulk data — IPC carries
|
||||||
@@ -206,6 +222,11 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
.signal_bind => systemSignalBind(state),
|
.signal_bind => systemSignalBind(state),
|
||||||
.process_signal => systemProcessSignal(state),
|
.process_signal => systemProcessSignal(state),
|
||||||
.timer_bind => systemTimerBind(state),
|
.timer_bind => systemTimerBind(state),
|
||||||
|
.klog_read => systemKlogRead(state),
|
||||||
|
.wall_clock => systemWallClock(state),
|
||||||
|
.shm_create => systemShmCreate(state),
|
||||||
|
.shm_map => systemShmMap(state),
|
||||||
|
.shm_physical => systemShmPhysical(state),
|
||||||
_ => fail(state),
|
_ => fail(state),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -298,12 +319,19 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||||
|
const device_id = architecture.systemCallArg(state, 0);
|
||||||
const claim_flags = sync.enter();
|
const claim_flags = sync.enter();
|
||||||
defer sync.leave(claim_flags);
|
defer sync.leave(claim_flags);
|
||||||
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
if (devices_broker.claim(device_id, scheduler.current().id)) {
|
||||||
architecture.setSystemCallResult(state, 0)
|
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||||
else
|
// so the kernel and the service don't scribble over each other's pixels. The
|
||||||
fail(state);
|
// claim releases (and the console resumes) automatically if the service dies;
|
||||||
|
// see releaseTaskResourcesLocked.
|
||||||
|
if (devices_broker.displayDevice()) |display_id| {
|
||||||
|
if (device_id == display_id) console.setSuppressed(true);
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
} else fail(state);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||||
@@ -338,7 +366,10 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
|||||||
const base_v = t.device_map_next;
|
const base_v = t.device_map_next;
|
||||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||||
|
|
||||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
|
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
|
||||||
|
// than the strong-uncacheable default that register MMIO needs.
|
||||||
|
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
|
||||||
|
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
|
||||||
t.device_map_next = base_v + pages * page_size;
|
t.device_map_next = base_v + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||||
}
|
}
|
||||||
@@ -444,6 +475,81 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||||
|
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
||||||
|
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
||||||
|
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
||||||
|
/// its frames are owned by a refcounted object: the handle is passed to another process as
|
||||||
|
/// an `ipc_call` send_cap, that process `shm_map`s it, and the frames free only when the
|
||||||
|
/// last capability drops (docs/display-v2.md — the compositor↔native-driver and
|
||||||
|
/// app↔compositor surface path).
|
||||||
|
fn systemShmCreate(state: *architecture.CpuState) void {
|
||||||
|
const len = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0 or len == 0) return fail(state);
|
||||||
|
|
||||||
|
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||||
|
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
||||||
|
|
||||||
|
// Reserve arena virtual space up front, so a mapping failure needs no rollback.
|
||||||
|
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||||
|
const base_v = t.shm_map_next;
|
||||||
|
if (base_v + pages * page_size > shm_arena_end) return fail(state); // arena exhausted
|
||||||
|
|
||||||
|
const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state);
|
||||||
|
// Zero through the physmap (the frames aren't mapped in the caller yet).
|
||||||
|
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||||
|
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||||
|
|
||||||
|
const shm = ipc.createShm(phys, pages) orelse {
|
||||||
|
for (0..pages) |i| pmm.free(phys + i * page_size);
|
||||||
|
return fail(state);
|
||||||
|
};
|
||||||
|
const handle = ipc.installShmHandle(t, shm);
|
||||||
|
if (handle < 0) {
|
||||||
|
ipc.dropShmRef(shm); // last ref: frees the object and its frames
|
||||||
|
return fail(state);
|
||||||
|
}
|
||||||
|
|
||||||
|
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
|
||||||
|
t.shm_map_next = base_v + pages * page_size;
|
||||||
|
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
|
||||||
|
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
|
||||||
|
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
||||||
|
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
||||||
|
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
||||||
|
fn systemShmMap(state: *architecture.CpuState) void {
|
||||||
|
const cap = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
|
||||||
|
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||||
|
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||||
|
const base_v = t.shm_map_next;
|
||||||
|
const size = shm.pages * page_size;
|
||||||
|
if (base_v + size > shm_arena_end) return fail(state);
|
||||||
|
|
||||||
|
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
|
||||||
|
t.shm_map_next = base_v + size;
|
||||||
|
architecture.setSystemCallResult(state, base_v);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a
|
||||||
|
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
||||||
|
/// physical base + length describes the whole region — which is exactly what a driver needs
|
||||||
|
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||||
|
/// capability can ask; there is no ambient way to turn a virtual address into a physical one.
|
||||||
|
fn systemShmPhysical(state: *architecture.CpuState) void {
|
||||||
|
const cap = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.aspace == 0) return fail(state);
|
||||||
|
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||||
|
architecture.setSystemCallResult(state, shm.phys);
|
||||||
|
}
|
||||||
|
|
||||||
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||||
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
||||||
/// enumerates it and hands each device it finds to the table, where a class driver
|
/// enumerates it and hands each device it finds to the table, where a class driver
|
||||||
@@ -598,6 +704,9 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
|||||||
recordExitLocked(t);
|
recordExitLocked(t);
|
||||||
irq.releaseOwner(t.id);
|
irq.releaseOwner(t.id);
|
||||||
devices_broker.releaseAllOwnedBy(t.id);
|
devices_broker.releaseAllOwnedBy(t.id);
|
||||||
|
// If that dropped the framebuffer claim (this task was the display service), let the
|
||||||
|
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
|
||||||
|
if (!devices_broker.displayClaimed()) console.setSuppressed(false);
|
||||||
// The dying task's signal endpoint and one-shot timers go with it.
|
// The dying task's signal endpoint and one-shot timers go with it.
|
||||||
if (t.signal_endpoint) |raw| {
|
if (t.signal_endpoint) |raw| {
|
||||||
ipc.dropRef(@ptrCast(@alignCast(raw)));
|
ipc.dropRef(@ptrCast(@alignCast(raw)));
|
||||||
@@ -628,6 +737,7 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
|||||||
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||||
}
|
}
|
||||||
ipc.abandonSenderLocked(t);
|
ipc.abandonSenderLocked(t);
|
||||||
|
ipc.killOwnedEndpointsLocked(t.id); // its registered services are gone: callers get -EPEER, not a hang
|
||||||
scheduler.removeFromWaitQueueLocked(t);
|
scheduler.removeFromWaitQueueLocked(t);
|
||||||
scheduler.forgetIpcClientLocked(t);
|
scheduler.forgetIpcClientLocked(t);
|
||||||
ipc.closeHandles(t);
|
ipc.closeHandles(t);
|
||||||
@@ -974,6 +1084,37 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// klog_read(offset, ptr, len) -> bytes copied: copy the kernel's in-memory
|
||||||
|
/// diagnostic log (the RAM sink in log.zig) out to the user buffer at `ptr`,
|
||||||
|
/// starting at `offset`. Returns the count copied — 0 once `offset` reaches the
|
||||||
|
/// end — so a program reads the whole log by looping from 0 until it gets 0.
|
||||||
|
///
|
||||||
|
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
||||||
|
/// but the copy runs kernel -> user. Written under the kernel lock so the source
|
||||||
|
/// snapshot can't grow underneath the copy. A read-only diagnostic — it exposes
|
||||||
|
/// only the log the kernel already broadcasts to serial, nothing else.
|
||||||
|
fn systemKlogRead(state: *architecture.CpuState) void {
|
||||||
|
const offset = architecture.systemCallArg(state, 0);
|
||||||
|
const ptr = architecture.systemCallArg(state, 1);
|
||||||
|
const len = architecture.systemCallArg(state, 2);
|
||||||
|
// Confine the whole destination span to the user (low) half. `len <=
|
||||||
|
// user_half_end - ptr` bounds the length without an overflowing add.
|
||||||
|
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
const snapshot = log.ramSnapshot();
|
||||||
|
var n: usize = 0;
|
||||||
|
if (offset < snapshot.len) {
|
||||||
|
n = @min(len, snapshot.len - offset);
|
||||||
|
const dest: [*]u8 = @ptrFromInt(ptr);
|
||||||
|
@memcpy(dest[0..n], snapshot[offset..][0..n]);
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, n);
|
||||||
|
} else {
|
||||||
|
fail(state);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||||
@@ -990,21 +1131,26 @@ fn systemMmap(state: *architecture.CpuState) void {
|
|||||||
const base = t.heap_next;
|
const base = t.heap_next;
|
||||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||||
|
|
||||||
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
// Map page by page. On mid-way frame exhaustion, roll back the pages already mapped
|
||||||
// (no partially-mapped grant leaks into the address space).
|
// (unmap + free) so no partial grant leaks into the address space — the same
|
||||||
var frames: [maximum_mmap_pages]u64 = undefined;
|
// all-or-nothing guarantee as before, but without a fixed scratch array, so the
|
||||||
var got: usize = 0;
|
// per-call size can be a multi-MiB framebuffer.
|
||||||
while (got < pages) : (got += 1) {
|
var mapped: usize = 0;
|
||||||
frames[got] = pmm.alloc() orelse {
|
while (mapped < pages) : (mapped += 1) {
|
||||||
for (frames[0..got]) |f| pmm.free(f);
|
const frame = pmm.alloc() orelse {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < mapped) : (i += 1) {
|
||||||
|
const va = base + i * page_size;
|
||||||
|
if (architecture.translate(t.aspace, va)) |physical| {
|
||||||
|
architecture.unmapUserPageInto(t.aspace, va);
|
||||||
|
pmm.free(physical);
|
||||||
|
}
|
||||||
|
}
|
||||||
return fail(state);
|
return fail(state);
|
||||||
};
|
};
|
||||||
}
|
|
||||||
|
|
||||||
for (frames[0..pages], 0..) |frame, i| {
|
|
||||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||||
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
|
||||||
}
|
}
|
||||||
t.heap_next = base + pages * page_size;
|
t.heap_next = base + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base);
|
architecture.setSystemCallResult(state, base);
|
||||||
@@ -1304,3 +1450,10 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
|||||||
fn systemClock(state: *architecture.CpuState) void {
|
fn systemClock(state: *architecture.CpuState) void {
|
||||||
architecture.setSystemCallResult(state, architecture.nanos());
|
architecture.setSystemCallResult(state, architecture.nanos());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// wall_clock() -> Unix epoch seconds (UTC). The RTC value, read at boot and offset
|
||||||
|
/// by the monotonic clock (wall-clock.zig) — mechanism, not policy: calendars and
|
||||||
|
/// timezones layer on top in user space. Needed for filesystem timestamps (mtime).
|
||||||
|
fn systemWallClock(state: *architecture.CpuState) void {
|
||||||
|
architecture.setSystemCallResult(state, wall_clock.nowSeconds());
|
||||||
|
}
|
||||||
|
|||||||
@@ -86,9 +86,11 @@ pub const Task = struct {
|
|||||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||||
device_map_next: u64 = 0,
|
device_map_next: u64 = 0,
|
||||||
// --- synchronous IPC (ipc_sync.zig) ---
|
// --- synchronous IPC (ipc_sync.zig) ---
|
||||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
// Per-process handle table: a small-int handle names a kernel capability object.
|
||||||
// opaque here so the scheduler and IPC modules don't import each other.
|
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
||||||
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
// close/exit and cap-passing paths reclaim the right type. Kept opaque here so the
|
||||||
|
// scheduler and IPC modules don't import each other (ipc_sync.zig owns the kinds).
|
||||||
|
handles: [ipc_maximum_handles]?HandleObject = .{null} ** ipc_maximum_handles,
|
||||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||||
@@ -99,6 +101,7 @@ pub const Task = struct {
|
|||||||
ipc_reply_cap: u64 = 0,
|
ipc_reply_cap: u64 = 0,
|
||||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||||
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
||||||
|
shm_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
|
||||||
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
||||||
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
||||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||||
@@ -125,6 +128,13 @@ pub const maximum_task_name = abi.maximum_process_name;
|
|||||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||||
pub const ipc_maximum_handles = 16;
|
pub const ipc_maximum_handles = 16;
|
||||||
|
|
||||||
|
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||||
|
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||||
|
/// capability-passing path reclaim/share the right type. The `kind` values are defined by
|
||||||
|
/// ipc_sync.zig (`handle_kind_*`); kept an opaque `u8` here so the scheduler doesn't import
|
||||||
|
/// the IPC module.
|
||||||
|
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||||
var next_id: u32 = 1;
|
var next_id: u32 = 1;
|
||||||
|
|
||||||
@@ -791,6 +801,21 @@ pub fn currentCpuIndex() u32 {
|
|||||||
return thisCpu().index;
|
return thisCpu().index;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The running task's id, or 0 if this core's scheduler isn't up yet (early boot, no GS
|
||||||
|
/// base). Safe for a fault reporter to call unconditionally — like `currentCpuIndex`,
|
||||||
|
/// it never dereferences an unpublished per-CPU pointer and so can't fault a second time.
|
||||||
|
pub fn currentIdSafe() u32 {
|
||||||
|
if (architecture.cpuLocal() == 0) return 0;
|
||||||
|
return thisCpu().current.id;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The running task's name (argv[0]), or "" if this core's scheduler isn't up yet.
|
||||||
|
/// The companion to `currentIdSafe` for naming the culprit in a fatal fault report.
|
||||||
|
pub fn currentNameSafe() []const u8 {
|
||||||
|
if (architecture.cpuLocal() == 0) return "";
|
||||||
|
return thisCpu().current.name();
|
||||||
|
}
|
||||||
|
|
||||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||||
pub fn setPriority(p: Priority) void {
|
pub fn setPriority(p: Priority) void {
|
||||||
current().priority = p;
|
current().priority = p;
|
||||||
|
|||||||
@@ -33,6 +33,13 @@ const architecture = @import("architecture");
|
|||||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||||
var held = std.atomic.Value(u32).init(0);
|
var held = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
/// The per-CPU base pointer (`architecture.cpuLocal()`) of the core currently holding
|
||||||
|
/// the lock, or 0 when free. Metadata only — `held` is what enforces exclusion — read
|
||||||
|
/// solely by `releaseIfHeldHere` on the fatal-fault path. `cpuLocal()` is a unique,
|
||||||
|
/// architecture-level token per core (0 before this core's GS base is published, which
|
||||||
|
/// is fine: that window is single-core early boot, where no other core can deadlock).
|
||||||
|
var owner = std.atomic.Value(usize).init(0);
|
||||||
|
|
||||||
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
||||||
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
||||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||||
@@ -67,14 +74,29 @@ export fn releaseForFreshTask() callconv(.c) void {
|
|||||||
release();
|
release();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Release the big kernel lock **only if this core is the one holding it** — a no-op
|
||||||
|
/// otherwise. For the fatal-fault path (a kernel-mode fault, or a nested fault inside the
|
||||||
|
/// recovery teardown, both of which run under the lock): a core that dies holding the BKL
|
||||||
|
/// must free it, or every other core spins forever in `acquire` and the whole machine
|
||||||
|
/// deadlocks instead of just that core stopping. It must NOT free a lock another core
|
||||||
|
/// owns, hence the owner check. Caveat: if we held it mid-mutation the shared state may be
|
||||||
|
/// inconsistent — but letting the other cores (and the supervisor) run on possibly-degraded
|
||||||
|
/// state is strictly more recoverable than a guaranteed total hang.
|
||||||
|
pub fn releaseIfHeldHere() void {
|
||||||
|
const me = architecture.cpuLocal();
|
||||||
|
if (me != 0 and owner.load(.monotonic) == me) release();
|
||||||
|
}
|
||||||
|
|
||||||
fn acquire() void {
|
fn acquire() void {
|
||||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||||
while (held.swap(1, .acquire) != 0) {
|
while (held.swap(1, .acquire) != 0) {
|
||||||
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||||
}
|
}
|
||||||
|
owner.store(architecture.cpuLocal(), .monotonic);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn release() void {
|
fn release() void {
|
||||||
|
owner.store(0, .monotonic);
|
||||||
held.store(0, .release);
|
held.store(0, .release);
|
||||||
}
|
}
|
||||||
|
|||||||
+638
-296
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,26 @@
|
|||||||
|
//! Wall-clock time: the CMOS real-time clock read once at boot and anchored to the
|
||||||
|
//! monotonic clock, so a query is a cheap arithmetic offset — no per-call CMOS poll,
|
||||||
|
//! no lock, no SMP hazard on the shared 0x70/0x71 ports.
|
||||||
|
//!
|
||||||
|
//! Wall-clock *seconds* are mechanism the kernel owns, exactly like the monotonic
|
||||||
|
//! clock ([[time-architecture]]): reading the hardware's value is not policy.
|
||||||
|
//! Calendars, timezones, and formatting layer on top in user space. It exists so the
|
||||||
|
//! filesystem can stamp real timestamps (mtime) — see docs/zig-self-hosting.md.
|
||||||
|
|
||||||
|
const architecture = @import("architecture");
|
||||||
|
|
||||||
|
var boot_unix_seconds: u64 = 0;
|
||||||
|
var boot_nanos: u64 = 0;
|
||||||
|
|
||||||
|
/// Read the RTC once and anchor it to the monotonic clock. Call at boot, after the
|
||||||
|
/// monotonic clock is calibrated.
|
||||||
|
pub fn init() void {
|
||||||
|
boot_unix_seconds = architecture.readRtcUnixSeconds();
|
||||||
|
boot_nanos = architecture.nanos();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The current wall-clock time in Unix epoch seconds (UTC): the boot RTC value plus
|
||||||
|
/// the monotonic time elapsed since. Zero until `init` runs.
|
||||||
|
pub fn nowSeconds() u64 {
|
||||||
|
return boot_unix_seconds + (architecture.nanos() -% boot_nanos) / 1_000_000_000;
|
||||||
|
}
|
||||||
@@ -17,10 +17,13 @@ pub const maximum_cpus = 128;
|
|||||||
|
|
||||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP. Sized
|
/// online core consumes one slot for its idle task, plus task 0 on the BSP. Sized
|
||||||
/// for the initial-ramdisk sweep (15 bundled binaries spawned at once) plus the
|
/// for the initial-ramdisk sweep (the bundled binaries spawned at once) plus the
|
||||||
/// device manager's supervised children with room to grow — at 16 the sweep
|
/// device manager's supervised children with room to grow — at 16 the sweep
|
||||||
/// started failing spawns once the bundle passed a dozen binaries.
|
/// started failing spawns once the bundle passed a dozen binaries. Raised to 48
|
||||||
pub const maximum_tasks = 32;
|
/// for the USB stack: the xHCI bus driver spawns a supervised class-driver instance
|
||||||
|
/// per matched interface (keyboard, mouse, mass storage), on top of the FAT and
|
||||||
|
/// block servers and the growing ramdisk bundle.
|
||||||
|
pub const maximum_tasks = 48;
|
||||||
|
|
||||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||||
pub const kernel_stack_size = 16 * 1024;
|
pub const kernel_stack_size = 16 * 1024;
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
||||||
//! interpreter, moved out of ring 0 (docs/m19-m20-plan.md, M20). Claims the
|
//! interpreter, moved out of ring 0 (docs/discovery.md). Claims the
|
||||||
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
||||||
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
||||||
//! ring 3 — the same parser and interpreter the kernel uses.
|
//! ring 3 — the same parser and interpreter the kernel uses.
|
||||||
@@ -103,21 +103,23 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: runtime.process.Init) void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||||
// own device count to self-verify against — deterministic, no log-scraping.
|
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||||
|
// devices is the check. Deterministic, no log-scraping.
|
||||||
|
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||||
|
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("acpi: out of memory\n");
|
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
const node = findTablesNode(buffer) orelse {
|
const node = findTablesNode(buffer) orelse {
|
||||||
_ = runtime.system.write("acpi: no acpi-tables node to claim\n");
|
_ = runtime.system.write("/system/services/acpi: no acpi-tables node to claim\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
node_id = node.id;
|
node_id = node.id;
|
||||||
if (!device.claim(node_id)) {
|
if (!device.claim(node_id)) {
|
||||||
_ = runtime.system.write("acpi: unable to claim acpi-tables\n");
|
_ = runtime.system.write("/system/services/acpi: unable to claim acpi-tables\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -151,22 +153,22 @@ pub fn main(init: runtime.process.Init) void {
|
|||||||
block_count += 1;
|
block_count += 1;
|
||||||
}
|
}
|
||||||
if (block_count == 0) {
|
if (block_count == 0) {
|
||||||
_ = runtime.system.write("acpi: no AML blobs on the node\n");
|
_ = runtime.system.write("/system/services/acpi: no AML blobs on the node\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
const result = aml.parse(runtime.allocator(), blocks[0..block_count]) catch {
|
const result = aml.parse(runtime.allocator(), blocks[0..block_count]) catch {
|
||||||
_ = runtime.system.write("acpi: AML parse failed\n");
|
_ = runtime.system.write("/system/services/acpi: AML parse failed\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
var namespace = result.namespace;
|
var namespace = result.namespace;
|
||||||
const devices = aml.deviceCount(&namespace);
|
const devices = aml.deviceCount(&namespace);
|
||||||
writeLine("acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||||
if (expected) |want| {
|
if (floor) |minimum| {
|
||||||
if (devices == want) {
|
if (devices >= minimum) {
|
||||||
_ = runtime.system.write("acpi-parse: ok\n");
|
_ = runtime.system.write("acpi-parse: ok\n");
|
||||||
} else {
|
} else {
|
||||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
writeLine("acpi-parse: too few (ring-3 {d} < floor {d})\n", .{ devices, minimum });
|
||||||
}
|
}
|
||||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||||
while (true) runtime.system.sleep(1000);
|
while (true) runtime.system.sleep(1000);
|
||||||
@@ -214,9 +216,9 @@ fn onInit(endpoint: runtime.ipc.Handle) bool {
|
|||||||
const hid = entry.hid[0..entry.hid_len];
|
const hid = entry.hid[0..entry.hid_len];
|
||||||
const desc = acpi_ids.description(hid);
|
const desc = acpi_ids.description(hid);
|
||||||
if (desc.len != 0)
|
if (desc.len != 0)
|
||||||
writeLine("acpi: reported {s} (device {d}, {d} resources) — {s}\n", .{ hid, entry.device_id, entry.resource_count, desc })
|
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources) — {s}\n", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||||
else
|
else
|
||||||
writeLine("acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
||||||
if (manager) |h| {
|
if (manager) |h| {
|
||||||
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||||
@@ -224,7 +226,7 @@ fn onInit(endpoint: runtime.ipc.Handle) bool {
|
|||||||
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
writeLine("acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||||
|
|
||||||
armPowerButton(endpoint);
|
armPowerButton(endpoint);
|
||||||
return true;
|
return true;
|
||||||
@@ -321,7 +323,7 @@ fn onSci() void {
|
|||||||
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
||||||
/// method produced, and publish an event per notified device. Then clear the
|
/// method produced, and publish an event per notified device. Then clear the
|
||||||
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
||||||
/// host unit tests (docs/m21-plan.md decision 5); on real hardware it carries
|
/// host unit tests (docs/acpi.md — ACPI events); on real hardware it carries
|
||||||
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
||||||
fn handleGpe() void {
|
fn handleGpe() void {
|
||||||
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
||||||
@@ -476,7 +478,7 @@ fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void {
|
|||||||
|
|
||||||
if (readHid(c, interpreter)) |hid| {
|
if (readHid(c, interpreter)) |hid| {
|
||||||
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
||||||
// only the non-PCI _HID devices (docs/m19-m20-plan.md M20.2). The two
|
// only the non-PCI _HID devices (docs/device-manager.md — matching). The two
|
||||||
// roots are named through the shared registry, not bare _HID strings.
|
// roots are named through the shared registry, not bare _HID strings.
|
||||||
const id = acpi_ids.HardwareId.fromHid(hid[0..7]);
|
const id = acpi_ids.HardwareId.fromHid(hid[0..7]);
|
||||||
if (id != .pci_bus and id != .pci_express_root_bridge) {
|
if (id != .pci_bus and id != .pci_express_root_bridge) {
|
||||||
@@ -498,7 +500,7 @@ fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) vo
|
|||||||
applyCrs(&descriptor, node, interpreter);
|
applyCrs(&descriptor, node, interpreter);
|
||||||
|
|
||||||
const id = device.register(node_id, &descriptor) orelse {
|
const id = device.register(node_id, &descriptor) orelse {
|
||||||
writeLine("acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
writeLine("/system/services/acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
//! The block-device wire protocol — what a filesystem (the FAT server) says to a
|
||||||
|
//! block driver (usb-storage) over its well-known `.block` endpoint. A protocol
|
||||||
|
//! module like vfs-protocol / usb-transfer-protocol: extern-struct messages, an
|
||||||
|
//! `Operation` tag, everything in one IPC message.
|
||||||
|
//!
|
||||||
|
//! Data path: read and write move whole blocks to or from a **caller-owned DMA
|
||||||
|
//! buffer**, named by its physical address — the same physical-address handoff
|
||||||
|
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||||
|
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||||
|
//! reply headers do. (Safe while the IOMMU is unenforced; see docs/driver-model.md.)
|
||||||
|
|
||||||
|
pub const Operation = enum(u32) {
|
||||||
|
/// geometry() -> { block_size, block_count }
|
||||||
|
geometry = 0,
|
||||||
|
/// read(lba, count, physical): read `count` blocks from `lba` into the buffer
|
||||||
|
read = 1,
|
||||||
|
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||||
|
write = 2,
|
||||||
|
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||||
|
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||||
|
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||||
|
flush = 3,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Request = extern struct {
|
||||||
|
operation: u32,
|
||||||
|
reserved: u32 = 0,
|
||||||
|
lba: u64,
|
||||||
|
count: u32, // number of blocks (read/write)
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
physical: u64, // caller's DMA buffer physical address (read/write)
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Reply = extern struct {
|
||||||
|
status: i32, // 0 on success, negative on failure
|
||||||
|
reserved: u32 = 0,
|
||||||
|
block_size: u32, // geometry: bytes per block (512)
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
block_count: u64, // geometry: total blocks; read/write: blocks moved
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const message_maximum: usize = 256;
|
||||||
|
pub const request_size: usize = @sizeOf(Request);
|
||||||
|
pub const reply_size: usize = @sizeOf(Reply);
|
||||||
@@ -19,6 +19,7 @@ const std = @import("std");
|
|||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
const acpi_ids = @import("acpi-ids");
|
const acpi_ids = @import("acpi-ids");
|
||||||
const pci_class = @import("pci-class");
|
const pci_class = @import("pci-class");
|
||||||
|
const usb_ids = @import("usb-ids");
|
||||||
const protocol = runtime.device_manager_protocol;
|
const protocol = runtime.device_manager_protocol;
|
||||||
const device = runtime.device;
|
const device = runtime.device;
|
||||||
const system = runtime.system;
|
const system = runtime.system;
|
||||||
@@ -31,17 +32,6 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
|||||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The driver that serves each device — the policy table. In a fuller system
|
|
||||||
/// this comes from a manifest (docs/device-manager.md: the third bus type
|
|
||||||
/// triggers it); for now a static map. `null` = no driver for this class yet.
|
|
||||||
fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
|
||||||
// The HPET timer node is still kernel-seeded (from the HPET table, not AML).
|
|
||||||
// PS/2 and other _HID devices now arrive as acpi-service reports and match
|
|
||||||
// in onChildAdded (M20.3), not from this boot snapshot.
|
|
||||||
if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||||
@@ -51,6 +41,15 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
|||||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||||
});
|
});
|
||||||
|
|
||||||
|
/// The PCI class triple of a virtio-gpu — Display Controller / Other (0x80) / 0. The class
|
||||||
|
/// alone cannot tell it from any other display/other function, so the driver re-confirms
|
||||||
|
/// vendor 0x1AF4 / device 0x1050 from config space once spawned; this only gets it spawned.
|
||||||
|
const virtio_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||||
|
.base = @intFromEnum(pci_class.BaseClass.display),
|
||||||
|
.subclass = 0x80, // "Other" — no named SubClass member (PCI convention)
|
||||||
|
.prog_if = 0,
|
||||||
|
});
|
||||||
|
|
||||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||||
/// several identical controllers — one driver instance per reported device,
|
/// several identical controllers — one driver instance per reported device,
|
||||||
@@ -58,6 +57,7 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
|||||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||||
return switch (identity) {
|
return switch (identity) {
|
||||||
xhci_pci_class => "usb-xhci-bus",
|
xhci_pci_class => "usb-xhci-bus",
|
||||||
|
virtio_gpu_pci_class => "virtio-gpu",
|
||||||
else => null,
|
else => null,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -72,6 +72,36 @@ fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The driver that serves a *reported* USB interface by its (class, subclass,
|
||||||
|
/// protocol) triple — the third bus after PCI and ACPI (docs/device-manager.md:
|
||||||
|
/// matching stays code until the third bus). The xHCI bus driver reports each
|
||||||
|
/// interface with this packed triple as its identity; the matched class driver is
|
||||||
|
/// spawned with the interface's registered id as argv[1], which it presents to the
|
||||||
|
/// bus driver to open the device.
|
||||||
|
fn usbDriverForIdentity(identity: u64) ?[]const u8 {
|
||||||
|
const keyboard = comptime usb_ids.packTriple(
|
||||||
|
@intFromEnum(usb_ids.Class.hid),
|
||||||
|
@intFromEnum(usb_ids.hid.SubClass.boot),
|
||||||
|
@intFromEnum(usb_ids.hid.Protocol.keyboard),
|
||||||
|
);
|
||||||
|
const mouse = comptime usb_ids.packTriple(
|
||||||
|
@intFromEnum(usb_ids.Class.hid),
|
||||||
|
@intFromEnum(usb_ids.hid.SubClass.boot),
|
||||||
|
@intFromEnum(usb_ids.hid.Protocol.mouse),
|
||||||
|
);
|
||||||
|
const storage = comptime usb_ids.packTriple(
|
||||||
|
@intFromEnum(usb_ids.Class.mass_storage),
|
||||||
|
@intFromEnum(usb_ids.mass_storage.SubClass.scsi),
|
||||||
|
@intFromEnum(usb_ids.mass_storage.Protocol.bulk_only),
|
||||||
|
);
|
||||||
|
return switch (identity) {
|
||||||
|
keyboard => "usb-hid-keyboard",
|
||||||
|
mouse => "usb-hid-mouse",
|
||||||
|
storage => "usb-storage",
|
||||||
|
else => null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
/// Whether some driver entry already serves registered device `device_id` —
|
/// Whether some driver entry already serves registered device `device_id` —
|
||||||
/// a re-report after a bus restart must not spawn a second instance.
|
/// a re-report after a bus restart must not spawn a second instance.
|
||||||
fn driverForDevice(device_id: u64) bool {
|
fn driverForDevice(device_id: u64) bool {
|
||||||
@@ -107,7 +137,7 @@ const Driver = struct {
|
|||||||
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||||
device_id: u64 = protocol.no_device,
|
device_id: u64 = protocol.no_device,
|
||||||
// Whether this driver speaks the protocol (hello expected, deadline
|
// Whether this driver speaks the protocol (hello expected, deadline
|
||||||
// enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted
|
// enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted
|
||||||
// but not yet required to hello.
|
// but not yet required to hello.
|
||||||
speaks_protocol: bool = false,
|
speaks_protocol: bool = false,
|
||||||
process_id: u32 = 0,
|
process_id: u32 = 0,
|
||||||
@@ -129,6 +159,8 @@ var test_restart_mode = false;
|
|||||||
var test_usb_restart_mode = false;
|
var test_usb_restart_mode = false;
|
||||||
var test_usb_killed = false;
|
var test_usb_killed = false;
|
||||||
var test_pci_restart_mode = false;
|
var test_pci_restart_mode = false;
|
||||||
|
var test_scanout_restart_mode = false;
|
||||||
|
var test_scanout_killed = false;
|
||||||
var test_kill_pid: u32 = 0;
|
var test_kill_pid: u32 = 0;
|
||||||
var test_kill_due_ns: u64 = 0;
|
var test_kill_due_ns: u64 = 0;
|
||||||
|
|
||||||
@@ -191,7 +223,7 @@ fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, report
|
|||||||
fn pruneChildrenOf(reporter: u32) void {
|
fn pruneChildrenOf(reporter: u32) void {
|
||||||
for (&children) |*child| {
|
for (&children) |*child| {
|
||||||
if (child.used and child.reporter == reporter) {
|
if (child.used and child.reporter == reporter) {
|
||||||
writeLine("device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||||
child.used = false;
|
child.used = false;
|
||||||
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
||||||
publishEvent(std.mem.asBytes(&event));
|
publishEvent(std.mem.asBytes(&event));
|
||||||
@@ -238,7 +270,7 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
|||||||
spawnDriver(driver);
|
spawnDriver(driver);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
writeLine("device-manager: driver table full; cannot supervise {s}\n", .{name});
|
writeLine("/system/services/device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
||||||
@@ -253,7 +285,7 @@ fn spawnDriver(driver: *Driver) void {
|
|||||||
argument_count = 1;
|
argument_count = 1;
|
||||||
}
|
}
|
||||||
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||||
writeLine("device-manager: failed to spawn {s}\n", .{driver.name()});
|
writeLine("/system/services/device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||||
driver.state = .failed;
|
driver.state = .failed;
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
@@ -267,9 +299,9 @@ fn spawnDriver(driver: *Driver) void {
|
|||||||
driver.state = .running;
|
driver.state = .running;
|
||||||
}
|
}
|
||||||
if (driver.device_id != protocol.no_device) {
|
if (driver.device_id != protocol.no_device) {
|
||||||
writeLine("device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
writeLine("/system/services/device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||||
} else {
|
} else {
|
||||||
writeLine("device-manager: spawned {s}\n", .{driver.name()});
|
writeLine("/system/services/device-manager: spawned {s}\n", .{driver.name()});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -281,7 +313,7 @@ fn onDriverExit(driver: *Driver) void {
|
|||||||
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
||||||
if (reason == .exited) {
|
if (reason == .exited) {
|
||||||
driver.state = .stopped;
|
driver.state = .stopped;
|
||||||
writeLine("device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
writeLine("/system/services/device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const now = system.clock();
|
const now = system.clock();
|
||||||
@@ -289,13 +321,13 @@ fn onDriverExit(driver: *Driver) void {
|
|||||||
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
||||||
if (driver.restarts >= crash_loop_cap) {
|
if (driver.restarts >= crash_loop_cap) {
|
||||||
driver.state = .failed;
|
driver.state = .failed;
|
||||||
writeLine("device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
writeLine("/system/services/device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
||||||
driver.state = .restarting;
|
driver.state = .restarting;
|
||||||
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
||||||
writeLine("device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
writeLine("/system/services/device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||||
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -306,7 +338,7 @@ fn onDriverExit(driver: *Driver) void {
|
|||||||
fn sweepDeadlines() void {
|
fn sweepDeadlines() void {
|
||||||
const now = system.clock();
|
const now = system.clock();
|
||||||
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
||||||
writeLine("device-manager: test mode: killing the reporter\n", .{});
|
writeLine("/system/services/device-manager: test mode: killing the reporter\n", .{});
|
||||||
_ = system.kill(test_kill_pid);
|
_ = system.kill(test_kill_pid);
|
||||||
test_kill_pid = 0;
|
test_kill_pid = 0;
|
||||||
}
|
}
|
||||||
@@ -314,7 +346,7 @@ fn sweepDeadlines() void {
|
|||||||
if (!driver.used) continue;
|
if (!driver.used) continue;
|
||||||
switch (driver.state) {
|
switch (driver.state) {
|
||||||
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
||||||
writeLine("device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
writeLine("/system/services/device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||||
_ = system.kill(driver.process_id);
|
_ = system.kill(driver.process_id);
|
||||||
// The exit notification finishes the job via onDriverExit.
|
// The exit notification finishes the job via onDriverExit.
|
||||||
},
|
},
|
||||||
@@ -331,7 +363,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
|
|
||||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||||
_ = runtime.system.write("device-manager: out of memory\n");
|
_ = runtime.system.write("/system/services/device-manager: out of memory\n");
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const total = device.enumerate(buffer);
|
const total = device.enumerate(buffer);
|
||||||
@@ -346,19 +378,15 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
addDriver("pci-bus", descriptor.id, true);
|
addDriver("pci-bus", descriptor.id, true);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// PCI functions no longer appear in the boot snapshot (M19.3): the
|
// Nothing else is matched from the boot snapshot today. The kernel-seeded
|
||||||
// pci-bus driver reports them, and onChildAdded matches from reports.
|
// HPET timer node is served by the kernel's own clock (docs/timers.md), not
|
||||||
const driver_name = driverFor(descriptor) orelse continue;
|
// a user-space driver; PCI functions and PS/2 _HID devices arrive later as
|
||||||
matched += 1;
|
// pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md).
|
||||||
// Skip a singleton that is already alive (the initial-ramdisk sweep test
|
// A fuller system's static class->driver manifest (docs/device-manager.md)
|
||||||
// starts every bundled binary bare, this manager included) — spawning a
|
// would slot in here.
|
||||||
// second instance would only lose the claim race and churn the log.
|
|
||||||
if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) {
|
|
||||||
addDriver(driver_name, protocol.no_device, false);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// The discovery service (docs/m19-m20-plan.md M20): one per firmware, packed
|
// The discovery service (docs/discovery.md): one per firmware, packed
|
||||||
// under the neutral name "discovery", spawned once at startup. It finds and
|
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||||
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||||
// match — it is the discoverer, not a driver bound to one device.
|
// match — it is the discoverer, not a driver bound to one device.
|
||||||
@@ -372,9 +400,9 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (matched == 0) {
|
if (matched == 0) {
|
||||||
_ = runtime.system.write("device-manager: no matchable devices\n");
|
_ = runtime.system.write("/system/services/device-manager: no matchable devices\n");
|
||||||
} else {
|
} else {
|
||||||
_ = runtime.system.write("device-manager: ok\n");
|
_ = runtime.system.write("/system/services/device-manager: ok\n");
|
||||||
}
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -395,13 +423,21 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
|||||||
var status: i32 = 0;
|
var status: i32 = 0;
|
||||||
if (hello.version != protocol.version) {
|
if (hello.version != protocol.version) {
|
||||||
status = -1;
|
status = -1;
|
||||||
writeLine("device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
writeLine("/system/services/device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||||
} else if (driverByProcess(sender)) |driver| {
|
} else if (driverByProcess(sender)) |driver| {
|
||||||
driver.state = .running;
|
driver.state = .running;
|
||||||
writeLine("device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||||
|
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||||
|
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||||
|
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "virtio-gpu")) {
|
||||||
|
test_scanout_killed = true;
|
||||||
|
test_kill_pid = sender;
|
||||||
|
test_kill_due_ns = system.clock() + 1_500_000_000;
|
||||||
|
_ = system.timerOnce(manager_endpoint, 1600);
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
status = -1;
|
status = -1;
|
||||||
writeLine("device-manager: hello from unknown process {d}\n", .{sender});
|
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||||
}
|
}
|
||||||
const hello_reply = protocol.HelloReply{ .status = status };
|
const hello_reply = protocol.HelloReply{ .status = status };
|
||||||
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
||||||
@@ -417,7 +453,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
var status: i32 = 0;
|
var status: i32 = 0;
|
||||||
if (driverByProcess(sender)) |driver| {
|
if (driverByProcess(sender)) |driver| {
|
||||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||||
writeLine("device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
writeLine("/system/services/device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||||
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
||||||
// Matching from reports (M19.3): a registered child whose identity
|
// Matching from reports (M19.3): a registered child whose identity
|
||||||
// names a driver gets one, once — re-reports after a bus restart
|
// names a driver gets one, once — re-reports after a bus restart
|
||||||
@@ -426,6 +462,11 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
if (pciDriverForIdentity(report.identity)) |child_driver| {
|
if (pciDriverForIdentity(report.identity)) |child_driver| {
|
||||||
if (!driverForDevice(report.device_id)) addDriver(child_driver, report.device_id, true);
|
if (!driverForDevice(report.device_id)) addDriver(child_driver, report.device_id, true);
|
||||||
}
|
}
|
||||||
|
// USB interface match: the reported identity is the packed class triple,
|
||||||
|
// and the class driver is spawned with the interface's registered id.
|
||||||
|
if (usbDriverForIdentity(report.identity)) |usb_driver| {
|
||||||
|
if (!driverForDevice(report.device_id)) addDriver(usb_driver, report.device_id, true);
|
||||||
|
}
|
||||||
// ACPI _HID match (M20.3): ps2-bus is a singleton that finds its own
|
// ACPI _HID match (M20.3): ps2-bus is a singleton that finds its own
|
||||||
// devices by hid, so spawn it once, without a device assignment.
|
// devices by hid, so spawn it once, without a device assignment.
|
||||||
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
||||||
@@ -478,7 +519,7 @@ fn onChildRemoved(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
var status: i32 = -1;
|
var status: i32 = -1;
|
||||||
for (&children) |*child| {
|
for (&children) |*child| {
|
||||||
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
||||||
writeLine("device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||||
child.used = false;
|
child.used = false;
|
||||||
status = 0;
|
status = 0;
|
||||||
}
|
}
|
||||||
@@ -536,6 +577,7 @@ pub fn main(init: runtime.process.Init) void {
|
|||||||
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
||||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||||
|
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||||
}
|
}
|
||||||
runtime.service.run(protocol.message_maximum, .{
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
.service = .device_manager,
|
.service = .device_manager,
|
||||||
|
|||||||
@@ -0,0 +1,92 @@
|
|||||||
|
//! system/services/display-demo — a hardware-free client of the display service, the
|
||||||
|
//! `input-source` analog for the compositor. It creates a wallpaper, a rectangle it moves
|
||||||
|
//! each frame, and a small cursor, then drives the compositor in a present loop — proof
|
||||||
|
//! that a *separate process* can compose a moving scene through the display service over
|
||||||
|
//! IPC, exercising the layer client API and damage-driven present end to end
|
||||||
|
//! (docs/display.md). It logs `display-demo: ok` once it has driven a run of frames.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const display = runtime.display;
|
||||||
|
const system = runtime.system;
|
||||||
|
const time = runtime.time;
|
||||||
|
const input = runtime.input;
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
const mode = display.info() orelse {
|
||||||
|
_ = system.write("display-demo: no display service\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// A full-screen wallpaper under everything.
|
||||||
|
const wallpaper = display.createLayer(0, 0, mode.width, mode.height, 0) orelse return createFailed();
|
||||||
|
_ = wallpaper.fill(0, 0, mode.width, mode.height, display.color(0x10, 0x18, 0x28));
|
||||||
|
|
||||||
|
// A rectangle that slides back and forth.
|
||||||
|
const box_w: u32 = 140;
|
||||||
|
const box_h: u32 = 100;
|
||||||
|
const box_y: i32 = 200;
|
||||||
|
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
||||||
|
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
||||||
|
|
||||||
|
// A little cursor on top. Its position is signed (the layer API is i32) and clamped to
|
||||||
|
// the screen; mouse motion arrives as relative deltas we accumulate below.
|
||||||
|
var cursor_x: i32 = @intCast(mode.width / 2);
|
||||||
|
var cursor_y: i32 = @intCast(mode.height / 2);
|
||||||
|
const cursor_max_x: i32 = @as(i32, @intCast(mode.width)) - 12;
|
||||||
|
const cursor_max_y: i32 = @as(i32, @intCast(mode.height)) - 12;
|
||||||
|
const cursor = display.createLayer(cursor_x, cursor_y, 12, 12, 2) orelse return createFailed();
|
||||||
|
_ = cursor.fill(0, 0, 12, 12, display.color(0xF0, 0xF0, 0xF0));
|
||||||
|
|
||||||
|
_ = display.present();
|
||||||
|
_ = system.write("display-demo: scene up; animating\n");
|
||||||
|
|
||||||
|
const span: i32 = @as(i32, @intCast(mode.width)) - @as(i32, @intCast(box_w));
|
||||||
|
var x: i32 = 0;
|
||||||
|
var dx: i32 = 8;
|
||||||
|
var frame: u32 = 0;
|
||||||
|
|
||||||
|
|
||||||
|
var mouse = input.subscribeMouse(); // type: ?input.MouseSubscriber
|
||||||
|
if (mouse == null) _ = system.write("display-demo: no mouse; animating without it\n");
|
||||||
|
|
||||||
|
while (true) : (frame += 1) {
|
||||||
|
if (mouse) |*ms| {
|
||||||
|
if (ms.next()) |event| {
|
||||||
|
cursor_x = clamp(cursor_x + event.dx, 0, cursor_max_x);
|
||||||
|
cursor_y = clamp(cursor_y + event.dy, 0, cursor_max_y);
|
||||||
|
_ = cursor.configure(cursor_x, cursor_y, 2, true);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
x += dx;
|
||||||
|
if (x <= 0) {
|
||||||
|
x = 0;
|
||||||
|
dx = -dx;
|
||||||
|
} else if (x >= span) {
|
||||||
|
x = span;
|
||||||
|
dx = -dx;
|
||||||
|
}
|
||||||
|
_ = box.configure(x, box_y, 1, true); // move it; the compositor repaints old + new
|
||||||
|
_ = display.present();
|
||||||
|
// A run of frames drawn through the compositor is the automated proof (the visible
|
||||||
|
// motion is a screenshot away via `zig build run-x86-64`).
|
||||||
|
if (frame == 20) _ = system.write("display-demo: ok\n");
|
||||||
|
time.sleep(time.Duration.fromMillis(30));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Clamp `v` to the inclusive range [lo, hi].
|
||||||
|
fn clamp(v: i32, lo: i32, hi: i32) i32 {
|
||||||
|
if (v < lo) return lo;
|
||||||
|
if (v > hi) return hi;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn createFailed() void {
|
||||||
|
_ = system.write("display-demo: create failed\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,273 @@
|
|||||||
|
//! The compositor's **scanout backend** — how a finished frame reaches the panel
|
||||||
|
//! (docs/display-v2.md). The compositor composes its layer stack into the backend's
|
||||||
|
//! cacheable `surface()` and calls `present(damage)`; everything device-specific lives
|
||||||
|
//! here. Today there is one backend, `Gop` — the firmware framebuffer: a cacheable back
|
||||||
|
//! buffer streamed write-combining to the linear framebuffer. A native virtio-gpu backend
|
||||||
|
//! slots in beside it later (V4); the compositor never learns which is active.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const compositor = @import("compositor.zig");
|
||||||
|
|
||||||
|
const system = runtime.system;
|
||||||
|
const device = runtime.device;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const scanout_protocol = runtime.scanout_protocol;
|
||||||
|
const Rect = compositor.Rect;
|
||||||
|
const Surface = compositor.Surface;
|
||||||
|
|
||||||
|
/// The current display mode, as a backend reports it.
|
||||||
|
pub const Info = struct { width: u32, height: u32, pitch: u32, format: u32 };
|
||||||
|
|
||||||
|
/// Enumeration scratch — a `DeviceDescriptor` is large, and only one scan is ever needed.
|
||||||
|
var device_table: [64]device.DeviceDescriptor = undefined;
|
||||||
|
|
||||||
|
/// The GOP framebuffer backend: claims the kernel-seeded `display` device, maps the linear
|
||||||
|
/// framebuffer write-combining as the front buffer, and keeps a cacheable back buffer of
|
||||||
|
/// the same geometry as the compose target. `present` streams the damaged rectangle from
|
||||||
|
/// the back buffer to the LFB (sequential WC writes; the LFB is never read). No mode-set,
|
||||||
|
/// no vsync — the portable floor (docs/display-v2.md).
|
||||||
|
pub const Gop = struct {
|
||||||
|
device_id: u64,
|
||||||
|
front: [*]volatile u8, // the LFB (write-combining)
|
||||||
|
back: [*]u8, // cacheable compose target, same geometry
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
pitch: u32,
|
||||||
|
format: u32,
|
||||||
|
|
||||||
|
/// The framebuffer's id and geometry, captured together. `findDisplay` reads these out of
|
||||||
|
/// the enumeration table and returns them by value, so the caller never re-reads the table
|
||||||
|
/// across later syscalls (`device_enumerate` writes the whole table straight into this
|
||||||
|
/// process's memory; reading a descriptor's tail again after other syscalls have run is a
|
||||||
|
/// window we simply avoid by copying the few fields we need up front).
|
||||||
|
const Found = struct { id: u64, width: u32, height: u32, pitch: u32, format: u32 };
|
||||||
|
|
||||||
|
/// The first `display`-class device with a *valid* (non-zero) geometry, or null. A zero
|
||||||
|
/// geometry is treated as "not ready yet" so the caller retries — a real framebuffer always
|
||||||
|
/// has a non-zero width, height, and pitch.
|
||||||
|
fn findDisplay() ?Found {
|
||||||
|
const total = device.enumerate(&device_table);
|
||||||
|
const n = @min(total, device_table.len);
|
||||||
|
for (device_table[0..n]) |*d| {
|
||||||
|
if (d.class != @intFromEnum(device.DeviceClass.display)) continue;
|
||||||
|
if (d.display.width == 0 or d.display.height == 0 or d.display.pitch == 0) continue;
|
||||||
|
return .{ .id = d.id, .width = d.display.width, .height = d.display.height, .pitch = d.display.pitch, .format = d.display.format };
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Claim the framebuffer (retrying while discovery catches up), map the LFB, and
|
||||||
|
/// allocate the back buffer. Null if there is no framebuffer or a mapping fails.
|
||||||
|
pub fn init() ?Gop {
|
||||||
|
var tries: u32 = 0;
|
||||||
|
const found = while (tries < 100) : (tries += 1) {
|
||||||
|
if (findDisplay()) |f| break f;
|
||||||
|
system.sleep(50);
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: no framebuffer device (headless?)\n");
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!device.claim(found.id)) {
|
||||||
|
_ = system.write("display: could not claim the framebuffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
// Resource 0 is the framebuffer memory window; the kernel maps it write-combining
|
||||||
|
// because the resource carries that flag (docs/display-plan.md D1).
|
||||||
|
const front_base = device.mmioMap(found.id, 0) orelse {
|
||||||
|
_ = system.write("display: could not map the framebuffer\n");
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
const size = @as(usize, found.height) * found.pitch;
|
||||||
|
const back_base = system.mmap(size, system.PROT_READ | system.PROT_WRITE);
|
||||||
|
if (system.mmapFailed(back_base)) {
|
||||||
|
_ = system.write("display: could not allocate the back buffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
return .{
|
||||||
|
.device_id = found.id,
|
||||||
|
.front = @ptrFromInt(front_base),
|
||||||
|
.back = @ptrFromInt(back_base),
|
||||||
|
.width = found.width,
|
||||||
|
.height = found.height,
|
||||||
|
.pitch = found.pitch,
|
||||||
|
.format = found.format,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn info(self: *const Gop) Info {
|
||||||
|
return .{ .width = self.width, .height = self.height, .pitch = self.pitch, .format = self.format };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cacheable compose target (the back buffer).
|
||||||
|
pub fn surface(self: *const Gop) Surface {
|
||||||
|
return .{
|
||||||
|
.pixels = @ptrCast(@alignCast(self.back)),
|
||||||
|
.stride = self.pitch / 4, // pitch is bytes; a 32-bpp row is pitch/4 pixels
|
||||||
|
.width = self.width,
|
||||||
|
.height = self.height,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stream the damaged rectangle from the back buffer to the write-combining LFB, row by
|
||||||
|
/// row (sequential writes — what WC memory wants; the LFB is never read).
|
||||||
|
pub fn present(self: *const Gop, damage: Rect) void {
|
||||||
|
const c = damage.intersect(.{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) });
|
||||||
|
if (c.isEmpty()) return;
|
||||||
|
var y: i32 = c.y;
|
||||||
|
while (y < c.bottom()) : (y += 1) {
|
||||||
|
const off = @as(usize, @intCast(y)) * self.pitch;
|
||||||
|
const src: [*]const u32 = @ptrCast(@alignCast(self.back + off));
|
||||||
|
const dst: [*]volatile u32 = @ptrCast(@alignCast(self.front + off));
|
||||||
|
var x: i32 = c.x;
|
||||||
|
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A display mode the native backend can switch to.
|
||||||
|
pub const Mode = scanout_protocol.Mode;
|
||||||
|
|
||||||
|
/// The native virtio-gpu backend: the compositor composes into a **shared** scanout surface
|
||||||
|
/// (an `shm` region the driver created and handed over) and `present` asks the driver to put
|
||||||
|
/// a frame on the panel over its `.scanout` endpoint. Unlike GOP there is no local copy — the
|
||||||
|
/// surface *is* the device's resource backing, so compositing writes land straight where the
|
||||||
|
/// driver transfers-and-flushes from (x86 DMA is cache-coherent, so the cacheable shared pages
|
||||||
|
/// need no explicit flush). Built by the display service when a driver announces (V4). The
|
||||||
|
/// surface is sized to the driver's largest mode, so `stride` (its row width) is fixed while
|
||||||
|
/// `width`/`height` — the active mode — change under `setMode` (V5).
|
||||||
|
pub const VirtioGpu = struct {
|
||||||
|
pixels: [*]u32, // the shared scanout surface, mapped into the compositor
|
||||||
|
stride: u32, // the surface's row stride in pixels (the driver's max mode width) — fixed
|
||||||
|
width: u32, // the active mode
|
||||||
|
height: u32,
|
||||||
|
format: u32,
|
||||||
|
scanout: ipc.Handle, // the driver's present + mode channel (looked up on `.scanout`)
|
||||||
|
|
||||||
|
pub fn info(self: *const VirtioGpu) Info {
|
||||||
|
return .{ .width = self.width, .height = self.height, .pitch = self.stride * 4, .format = self.format };
|
||||||
|
}
|
||||||
|
pub fn surface(self: *const VirtioGpu) Surface {
|
||||||
|
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
|
||||||
|
}
|
||||||
|
/// Ask the driver to present. The composited pixels are already in the shared surface, so
|
||||||
|
/// this is a single request over `.scanout`; the driver transfers + fenced-flushes.
|
||||||
|
pub fn present(self: *const VirtioGpu, damage: Rect) void {
|
||||||
|
_ = damage;
|
||||||
|
var request = scanout_protocol.Request{
|
||||||
|
.operation = @intFromEnum(scanout_protocol.Operation.present),
|
||||||
|
.width = self.width,
|
||||||
|
.height = self.height,
|
||||||
|
};
|
||||||
|
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||||
|
_ = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch {};
|
||||||
|
}
|
||||||
|
/// Fill `out` with the driver's offered modes; returns how many were written.
|
||||||
|
pub fn modes(self: *const VirtioGpu, out: []Mode) usize {
|
||||||
|
var request = scanout_protocol.Request{ .operation = @intFromEnum(scanout_protocol.Operation.get_modes) };
|
||||||
|
var reply: [scanout_protocol.modes_reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return 0;
|
||||||
|
if (n < scanout_protocol.modes_reply_size) return 0;
|
||||||
|
const answer = std.mem.bytesToValue(scanout_protocol.ModesReply, reply[0..scanout_protocol.modes_reply_size]);
|
||||||
|
if (answer.status != 0) return 0;
|
||||||
|
const count = @min(@min(answer.count, scanout_protocol.max_modes), out.len);
|
||||||
|
for (0..count) |i| out[i] = answer.modes[i];
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
/// Change the scanout resolution. On success the active `width`/`height` update (the shared
|
||||||
|
/// surface — sized to the max mode — is unchanged, so `stride` stays put).
|
||||||
|
pub fn setMode(self: *VirtioGpu, w: u32, h: u32) bool {
|
||||||
|
if (w == 0 or h == 0 or w > self.stride) return false;
|
||||||
|
var request = scanout_protocol.Request{
|
||||||
|
.operation = @intFromEnum(scanout_protocol.Operation.set_mode),
|
||||||
|
.width = w,
|
||||||
|
.height = h,
|
||||||
|
};
|
||||||
|
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (n < scanout_protocol.reply_size) return false;
|
||||||
|
if (std.mem.bytesToValue(scanout_protocol.Reply, reply[0..scanout_protocol.reply_size]).status != 0) return false;
|
||||||
|
self.width = w;
|
||||||
|
self.height = h;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The pluggable scanout backend. A tagged union so the compositor holds one value and
|
||||||
|
/// dispatches without caring which is active; the `virtio` native backend joins `gop` at V4.
|
||||||
|
pub const Backend = union(enum) {
|
||||||
|
gop: Gop,
|
||||||
|
virtio: VirtioGpu,
|
||||||
|
|
||||||
|
pub fn info(self: *const Backend) Info {
|
||||||
|
return switch (self.*) {
|
||||||
|
inline else => |*b| b.info(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
pub fn surface(self: *const Backend) Surface {
|
||||||
|
return switch (self.*) {
|
||||||
|
inline else => |*b| b.surface(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
pub fn present(self: *const Backend, damage: Rect) void {
|
||||||
|
switch (self.*) {
|
||||||
|
inline else => |*b| b.present(damage),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// The modes this backend can switch to (none for GOP); returns how many were written.
|
||||||
|
pub fn modes(self: *const Backend, out: []Mode) usize {
|
||||||
|
return switch (self.*) {
|
||||||
|
.virtio => |*v| v.modes(out),
|
||||||
|
.gop => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
/// Change the resolution; false if this backend can't mode-set or the mode was refused.
|
||||||
|
pub fn setMode(self: *Backend, w: u32, h: u32) bool {
|
||||||
|
return switch (self.*) {
|
||||||
|
.virtio => |*v| v.setMode(w, h),
|
||||||
|
.gop => false,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
/// Whether this backend supports runtime mode-setting (GOP: no; virtio-gpu: yes, V5).
|
||||||
|
pub fn canModeSet(self: *const Backend) bool {
|
||||||
|
return switch (self.*) {
|
||||||
|
.gop => false,
|
||||||
|
.virtio => true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
/// Whether this backend has a vblank/fence for tear-free present (virtio-gpu: yes, V5 — every
|
||||||
|
/// flush is fenced, so the device signals completion when the frame is actually on screen).
|
||||||
|
pub fn hasVsync(self: *const Backend) bool {
|
||||||
|
return switch (self.*) {
|
||||||
|
.gop => false,
|
||||||
|
.virtio => true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Which backend to use. The pure selection *decision* is `chooseKind`; `select` below
|
||||||
|
/// binds it to the (syscall-bound) bring-up.
|
||||||
|
pub const Kind = enum { gop, virtio };
|
||||||
|
|
||||||
|
/// The selection decision, factored out of bring-up so it stays pure and host-testable:
|
||||||
|
/// prefer a native driver when one has announced itself (docs/display-v2.md V4), else the
|
||||||
|
/// GOP floor. Trivial today; it grows real inputs when native detection lands.
|
||||||
|
pub fn chooseKind(native_available: bool) Kind {
|
||||||
|
return if (native_available) .virtio else .gop;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pick and bring up the best available backend. Today the GOP framebuffer is the only one
|
||||||
|
/// (`chooseKind(false)` → `.gop`), so this is `Gop.init()`. V4 adds the native-if-present
|
||||||
|
/// branch, with GOP as the floor.
|
||||||
|
pub fn select() ?Backend {
|
||||||
|
return switch (chooseKind(false)) {
|
||||||
|
.gop => .{ .gop = Gop.init() orelse return null },
|
||||||
|
.virtio => unreachable, // no native detection yet (V4)
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "selection prefers native when present, else the gop floor" {
|
||||||
|
try std.testing.expectEqual(Kind.gop, chooseKind(false));
|
||||||
|
try std.testing.expectEqual(Kind.virtio, chooseKind(true));
|
||||||
|
}
|
||||||
@@ -0,0 +1,192 @@
|
|||||||
|
//! The compositor's pure core: rectangle math and the three blitting primitives the
|
||||||
|
//! display service composes frames from — fill a rectangle of a surface, composite one
|
||||||
|
//! surface onto another clipped to a damage rectangle, and copy a client-supplied pixel
|
||||||
|
//! tile in. Deliberately free of any syscall or `runtime` dependency (it takes plain
|
||||||
|
//! pixel pointers), so it is host-tested under `zig build test`. The service
|
||||||
|
//! (system/services/display/display.zig) wires real mmap'd surfaces and the framebuffer
|
||||||
|
//! to it. Pixels are opaque native 32-bit values — v1 layers don't alpha-blend, and
|
||||||
|
//! channel order (rgbx/bgrx) is the caller's concern (see protocol.pack).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// An axis-aligned rectangle in pixels. Signed, so a surface partly off-screen (a layer
|
||||||
|
/// dragged past an edge) clips with plain arithmetic. Half-open: covers [x, x+w) × [y, y+h).
|
||||||
|
pub const Rect = struct {
|
||||||
|
x: i32,
|
||||||
|
y: i32,
|
||||||
|
w: i32,
|
||||||
|
h: i32,
|
||||||
|
|
||||||
|
pub const empty = Rect{ .x = 0, .y = 0, .w = 0, .h = 0 };
|
||||||
|
|
||||||
|
pub fn init(x: i32, y: i32, w: i32, h: i32) Rect {
|
||||||
|
return .{ .x = x, .y = y, .w = w, .h = h };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn isEmpty(r: Rect) bool {
|
||||||
|
return r.w <= 0 or r.h <= 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn right(r: Rect) i32 {
|
||||||
|
return r.x + r.w;
|
||||||
|
}
|
||||||
|
pub fn bottom(r: Rect) i32 {
|
||||||
|
return r.y + r.h;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The overlap of two rectangles, or an empty rectangle if they don't touch.
|
||||||
|
pub fn intersect(a: Rect, b: Rect) Rect {
|
||||||
|
const x0 = @max(a.x, b.x);
|
||||||
|
const y0 = @max(a.y, b.y);
|
||||||
|
const x1 = @min(a.right(), b.right());
|
||||||
|
const y1 = @min(a.bottom(), b.bottom());
|
||||||
|
return .{ .x = x0, .y = y0, .w = x1 - x0, .h = y1 - y0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bounding box of two rectangles. An empty operand contributes nothing (returns
|
||||||
|
/// the other), so folding damage rectangles with `unite` from `empty` yields their
|
||||||
|
/// bounding box.
|
||||||
|
pub fn unite(a: Rect, b: Rect) Rect {
|
||||||
|
if (a.isEmpty()) return b;
|
||||||
|
if (b.isEmpty()) return a;
|
||||||
|
const x0 = @min(a.x, b.x);
|
||||||
|
const y0 = @min(a.y, b.y);
|
||||||
|
const x1 = @max(a.right(), b.right());
|
||||||
|
const y1 = @max(a.bottom(), b.bottom());
|
||||||
|
return .{ .x = x0, .y = y0, .w = x1 - x0, .h = y1 - y0 };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A block of 32-bit pixels: `pixels` addressed row-major with `stride` pixels between
|
||||||
|
/// row starts (≥ width — the framebuffer's stride is pitch/4, a layer's is its width).
|
||||||
|
pub const Surface = struct {
|
||||||
|
pixels: [*]u32,
|
||||||
|
stride: u32, // pixels per row
|
||||||
|
width: u32,
|
||||||
|
height: u32,
|
||||||
|
|
||||||
|
pub fn bounds(s: Surface) Rect {
|
||||||
|
return .{ .x = 0, .y = 0, .w = @intCast(s.width), .h = @intCast(s.height) };
|
||||||
|
}
|
||||||
|
|
||||||
|
inline fn row(s: Surface, y: u32) [*]u32 {
|
||||||
|
return s.pixels + @as(usize, y) * s.stride;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Fill `rect` of `s` with the native pixel `colour`, clipped to `s`'s bounds.
|
||||||
|
pub fn fillRect(s: Surface, rect: Rect, colour: u32) void {
|
||||||
|
const c = rect.intersect(s.bounds());
|
||||||
|
if (c.isEmpty()) return;
|
||||||
|
var y: i32 = c.y;
|
||||||
|
while (y < c.bottom()) : (y += 1) {
|
||||||
|
const r = s.row(@intCast(y));
|
||||||
|
var x: i32 = c.x;
|
||||||
|
while (x < c.right()) : (x += 1) r[@intCast(x)] = colour;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Composite the whole of `layer` onto `dst` with the layer's top-left at (`dx`, `dy`),
|
||||||
|
/// painting only the pixels that fall inside `clip` (a `dst`-space rectangle) and inside
|
||||||
|
/// `dst`. Opaque copy. This is the primitive `present` repeats over the visible layer
|
||||||
|
/// stack, bottom to top, for each damaged region.
|
||||||
|
pub fn composite(dst: Surface, dx: i32, dy: i32, layer: Surface, clip: Rect) void {
|
||||||
|
const on_screen = Rect{ .x = dx, .y = dy, .w = @intCast(layer.width), .h = @intCast(layer.height) };
|
||||||
|
const region = on_screen.intersect(clip).intersect(dst.bounds());
|
||||||
|
if (region.isEmpty()) return;
|
||||||
|
var y: i32 = region.y;
|
||||||
|
while (y < region.bottom()) : (y += 1) {
|
||||||
|
const src = layer.row(@intCast(y - dy));
|
||||||
|
const d = dst.row(@intCast(y));
|
||||||
|
var x: i32 = region.x;
|
||||||
|
while (x < region.right()) : (x += 1) {
|
||||||
|
d[@intCast(x)] = src[@intCast(x - dx)];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy a `w`×`h` tile of native pixels from `src` (raw little-endian bytes, row-major,
|
||||||
|
/// tightly packed) into `dst` at (`dx`, `dy`), clipped to `dst`'s bounds. `src` is read
|
||||||
|
/// with `readInt` because it comes straight out of an IPC message buffer and carries no
|
||||||
|
/// alignment guarantee. Returns without touching anything if `src` is short.
|
||||||
|
pub fn blitTile(dst: Surface, dx: i32, dy: i32, src: []const u8, w: u32, h: u32) void {
|
||||||
|
if (src.len < @as(usize, w) * h * 4) return;
|
||||||
|
var ty: u32 = 0;
|
||||||
|
while (ty < h) : (ty += 1) {
|
||||||
|
const yy = dy + @as(i32, @intCast(ty));
|
||||||
|
if (yy < 0 or yy >= dst.height) continue;
|
||||||
|
const drow = dst.row(@intCast(yy));
|
||||||
|
var tx: u32 = 0;
|
||||||
|
while (tx < w) : (tx += 1) {
|
||||||
|
const xx = dx + @as(i32, @intCast(tx));
|
||||||
|
if (xx < 0 or xx >= dst.width) continue;
|
||||||
|
const off = (@as(usize, ty) * w + tx) * 4;
|
||||||
|
drow[@intCast(xx)] = std.mem.readInt(u32, src[off..][0..4], .little);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
|
test "rect intersect: overlap and disjoint" {
|
||||||
|
try std.testing.expectEqual(Rect.init(5, 5, 5, 5), Rect.init(0, 0, 10, 10).intersect(Rect.init(5, 5, 10, 10)));
|
||||||
|
try std.testing.expect(Rect.init(0, 0, 10, 10).intersect(Rect.init(20, 20, 5, 5)).isEmpty());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "rect unite: bounding box, empty is identity" {
|
||||||
|
const a = Rect.init(2, 2, 4, 4);
|
||||||
|
try std.testing.expectEqual(Rect.init(2, 1, 10, 5), a.unite(Rect.init(10, 1, 2, 2)));
|
||||||
|
try std.testing.expectEqual(a, a.unite(Rect.empty));
|
||||||
|
try std.testing.expectEqual(a, Rect.empty.unite(a));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "fillRect clips to surface and honours stride padding" {
|
||||||
|
// A 4×3 surface inside a 6-wide allocation (stride 6 > width 4), like pitch padding.
|
||||||
|
var mem = [_]u32{0} ** (6 * 3);
|
||||||
|
const s = Surface{ .pixels = &mem, .stride = 6, .width = 4, .height = 3 };
|
||||||
|
fillRect(s, Rect.init(-1, -1, 3, 3), 0xAB); // straddles the top-left corner
|
||||||
|
try std.testing.expectEqual(@as(u32, 0xAB), mem[0 * 6 + 0]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0xAB), mem[1 * 6 + 1]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), mem[0 * 6 + 2]); // beyond the 2-wide fill
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), mem[2 * 6 + 0]); // row 2 untouched
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), mem[0 * 6 + 4]); // stride padding untouched
|
||||||
|
}
|
||||||
|
|
||||||
|
test "composite: overlap shows the top layer, clipped to damage" {
|
||||||
|
var back = [_]u32{0} ** (8 * 8);
|
||||||
|
const dst = Surface{ .pixels = &back, .stride = 8, .width = 8, .height = 8 };
|
||||||
|
var lo = [_]u32{0x11} ** (4 * 4);
|
||||||
|
var hi = [_]u32{0x22} ** (4 * 4);
|
||||||
|
const low = Surface{ .pixels = &lo, .stride = 4, .width = 4, .height = 4 };
|
||||||
|
const high = Surface{ .pixels = &hi, .stride = 4, .width = 4, .height = 4 };
|
||||||
|
composite(dst, 0, 0, low, dst.bounds()); // bottom at (0,0)
|
||||||
|
composite(dst, 2, 2, high, dst.bounds()); // top overlaps at (2,2)
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x11), back[0 * 8 + 0]); // bottom-only
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x22), back[3 * 8 + 3]); // overlap → top wins
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x22), back[5 * 8 + 5]); // top-only
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[7 * 8 + 7]); // neither
|
||||||
|
}
|
||||||
|
|
||||||
|
test "composite honours the damage rectangle" {
|
||||||
|
var back = [_]u32{0} ** (8 * 8);
|
||||||
|
const dst = Surface{ .pixels = &back, .stride = 8, .width = 8, .height = 8 };
|
||||||
|
var fill = [_]u32{0x33} ** (8 * 8);
|
||||||
|
const layer = Surface{ .pixels = &fill, .stride = 8, .width = 8, .height = 8 };
|
||||||
|
composite(dst, 0, 0, layer, Rect.init(2, 2, 2, 2)); // only this damage region
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x33), back[2 * 8 + 2]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x33), back[3 * 8 + 3]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[1 * 8 + 1]); // outside damage
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[4 * 8 + 4]); // outside damage
|
||||||
|
}
|
||||||
|
|
||||||
|
test "blitTile copies a packed tile, clipping and reading unaligned bytes" {
|
||||||
|
var back = [_]u32{0} ** (4 * 4);
|
||||||
|
const dst = Surface{ .pixels = &back, .stride = 4, .width = 4, .height = 4 };
|
||||||
|
// A 2×2 tile in a byte buffer offset by one byte, so reads are unaligned.
|
||||||
|
var raw = [_]u8{0} ** (1 + 2 * 2 * 4);
|
||||||
|
const tile = raw[1..];
|
||||||
|
for (0..4) |i| std.mem.writeInt(u32, tile[i * 4 ..][0..4], @intCast(0xA0 + i), .little);
|
||||||
|
blitTile(dst, 3, 3, tile, 2, 2); // bottom-right corner; only (3,3) lands on-surface
|
||||||
|
try std.testing.expectEqual(@as(u32, 0xA0), back[3 * 4 + 3]);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), back[0]); // nothing else touched
|
||||||
|
}
|
||||||
@@ -0,0 +1,462 @@
|
|||||||
|
//! /system/services/display — the display service (docs/display.md, docs/display-v2.md).
|
||||||
|
//! A ring-3 compositor: it composes an ordered stack of **layers** into a cacheable
|
||||||
|
//! surface and presents finished frames. Scanout — how a frame reaches the panel — is a
|
||||||
|
//! pluggable **backend** ([backend.zig](backend.zig)): the GOP framebuffer today, a native
|
||||||
|
//! virtio-gpu driver later; this file never learns which is active. It owns the layer stack
|
||||||
|
//! and damage tracking; the pixel math is the pure, host-tested
|
||||||
|
//! [compositor.zig](compositor.zig).
|
||||||
|
//!
|
||||||
|
//! A layer is a server-owned surface (its own cacheable buffer) with a screen position,
|
||||||
|
//! z-order, and visibility. Clients create layers, draw into them by command (`fill_rect`,
|
||||||
|
//! `blit_tile`), mark `damage`, and ask for a `present`; the compositor repaints only the
|
||||||
|
//! damaged region — clear it, paint the visible layers bottom-to-top into the backend's
|
||||||
|
//! surface, then `backend.present(damage)`. Shared-memory client surfaces are later
|
||||||
|
//! (docs/display-v2.md).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const compositor = @import("compositor.zig");
|
||||||
|
const backend_mod = @import("backend.zig");
|
||||||
|
|
||||||
|
const protocol = runtime.display_protocol;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
const system = runtime.system;
|
||||||
|
const Rect = compositor.Rect;
|
||||||
|
const Surface = compositor.Surface;
|
||||||
|
|
||||||
|
/// The active scanout backend — the GOP framebuffer at boot, upgraded to a native driver
|
||||||
|
/// (virtio-gpu) when one announces itself (V4).
|
||||||
|
var backend: backend_mod.Backend = undefined;
|
||||||
|
var frames: u64 = 0;
|
||||||
|
|
||||||
|
/// This service's endpoint, kept so `attach_scanout` can arm a one-shot timer: the very first
|
||||||
|
/// native present must happen in a *later* loop iteration, after the reply to the driver's
|
||||||
|
/// announce has unblocked it and it is serving its `.scanout` channel — presenting inline
|
||||||
|
/// would deadlock (we'd call the driver while it waits on our reply).
|
||||||
|
var service_endpoint: ipc.Handle = 0;
|
||||||
|
|
||||||
|
/// Set when the backend has just been upgraded to virtio-gpu: the next present repaints the
|
||||||
|
/// whole screen into the shared surface and reads a pixel back to confirm the frame landed.
|
||||||
|
var pending_native_verify: bool = false;
|
||||||
|
|
||||||
|
/// Set alongside it: after the native present is verified, run the mode-set self-check once
|
||||||
|
/// (query the driver's modes, switch to a different one, confirm the geometry changed) — the
|
||||||
|
/// serial proof the runtime-resolution-change + fenced-present paths work (V5).
|
||||||
|
var pending_modeset_check: bool = false;
|
||||||
|
|
||||||
|
/// The wallpaper the compositor clears damaged regions to before painting layers.
|
||||||
|
var background: u32 = 0;
|
||||||
|
|
||||||
|
/// The layer stack. A fixed table (a compositor has few top-level surfaces during
|
||||||
|
/// bring-up); each used slot owns an mmap'd surface. `damage` accumulates the dirty
|
||||||
|
/// screen region since the last `present`, so a present touches only what changed.
|
||||||
|
const maximum_layers = 16;
|
||||||
|
|
||||||
|
const Layer = struct {
|
||||||
|
used: bool = false,
|
||||||
|
x: i32 = 0,
|
||||||
|
y: i32 = 0,
|
||||||
|
z: u32 = 0,
|
||||||
|
visible: bool = false,
|
||||||
|
surface: Surface = undefined,
|
||||||
|
surface_len: usize = 0, // for munmap on destroy
|
||||||
|
};
|
||||||
|
|
||||||
|
var layers: [maximum_layers]Layer = [_]Layer{.{}} ** maximum_layers;
|
||||||
|
var damage: Rect = Rect.empty;
|
||||||
|
|
||||||
|
// --- geometry helpers -------------------------------------------------------
|
||||||
|
|
||||||
|
fn screenRect() Rect {
|
||||||
|
const m = backend.info();
|
||||||
|
return .{ .x = 0, .y = 0, .w = @intCast(m.width), .h = @intCast(m.height) };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn layerScreenRect(l: *const Layer) Rect {
|
||||||
|
return .{ .x = l.x, .y = l.y, .w = @intCast(l.surface.width), .h = @intCast(l.surface.height) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Add `r` (screen coordinates) to the pending damage, clipped to the screen.
|
||||||
|
fn addDamage(r: Rect) void {
|
||||||
|
damage = damage.unite(r.intersect(screenRect()));
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- layer operations (called from onMessage and the self-check) ------------
|
||||||
|
|
||||||
|
fn freeLayer() ?u32 {
|
||||||
|
for (&layers, 0..) |*l, i| {
|
||||||
|
if (!l.used) return @intCast(i);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A used layer by id, or null if the id is out of range or free.
|
||||||
|
fn layerAt(id: u32) ?*Layer {
|
||||||
|
if (id >= maximum_layers or !layers[id].used) return null;
|
||||||
|
return &layers[id];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32, visible: bool) ?u32 {
|
||||||
|
if (w == 0 or h == 0) return null;
|
||||||
|
const slot = freeLayer() orelse return null;
|
||||||
|
const len = @as(usize, w) * h * 4;
|
||||||
|
const base = system.mmap(len, system.PROT_READ | system.PROT_WRITE);
|
||||||
|
if (system.mmapFailed(base)) return null;
|
||||||
|
layers[slot] = .{
|
||||||
|
.used = true,
|
||||||
|
.x = x,
|
||||||
|
.y = y,
|
||||||
|
.z = z,
|
||||||
|
.visible = visible,
|
||||||
|
.surface = .{ .pixels = @ptrFromInt(base), .stride = w, .width = w, .height = h },
|
||||||
|
.surface_len = len,
|
||||||
|
};
|
||||||
|
return slot;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fillLayer(id: u32, local: Rect, colour: u32) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
compositor.fillRect(l.surface, local, colour);
|
||||||
|
// Damage in screen space = the fill, translated by the layer origin, within the layer.
|
||||||
|
const screen = Rect{ .x = l.x + local.x, .y = l.y + local.y, .w = local.w, .h = local.h };
|
||||||
|
addDamage(screen.intersect(layerScreenRect(l)));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn blitLayer(id: u32, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
compositor.blitTile(l.surface, x, y, pixels, w, h);
|
||||||
|
const screen = Rect{ .x = l.x + x, .y = l.y + y, .w = @intCast(w), .h = @intCast(h) };
|
||||||
|
addDamage(screen.intersect(layerScreenRect(l)));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn configureLayer(id: u32, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
addDamage(layerScreenRect(l)); // the old footprint must repaint
|
||||||
|
l.x = x;
|
||||||
|
l.y = y;
|
||||||
|
l.z = z;
|
||||||
|
l.visible = visible;
|
||||||
|
addDamage(layerScreenRect(l)); // and the new one
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn destroyLayer(id: u32) bool {
|
||||||
|
const l = layerAt(id) orelse return false;
|
||||||
|
addDamage(layerScreenRect(l));
|
||||||
|
_ = system.munmap(@intFromPtr(l.surface.pixels), l.surface_len);
|
||||||
|
l.* = .{};
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- compositing + present --------------------------------------------------
|
||||||
|
|
||||||
|
/// Repaint the damaged region `clip` of the backend's compose surface: clear it to the
|
||||||
|
/// background, then paint every visible layer that overlaps it, bottom to top (ascending z).
|
||||||
|
fn compositeInto(clip: Rect) void {
|
||||||
|
const target = backend.surface();
|
||||||
|
compositor.fillRect(target, clip, background);
|
||||||
|
|
||||||
|
// z-order the used, visible layers (n ≤ 16; a plain insertion sort of indices).
|
||||||
|
var order: [maximum_layers]u32 = undefined;
|
||||||
|
var n: usize = 0;
|
||||||
|
for (layers, 0..) |l, i| {
|
||||||
|
if (l.used and l.visible) {
|
||||||
|
order[n] = @intCast(i);
|
||||||
|
n += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var a: usize = 1;
|
||||||
|
while (a < n) : (a += 1) {
|
||||||
|
const key = order[a];
|
||||||
|
var b: usize = a;
|
||||||
|
while (b > 0 and layers[order[b - 1]].z > layers[key].z) : (b -= 1) order[b] = order[b - 1];
|
||||||
|
order[b] = key;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (order[0..n]) |i| {
|
||||||
|
const l = layers[i];
|
||||||
|
compositor.composite(target, l.x, l.y, l.surface, clip);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Composite the accumulated damage into the backend's surface, hand it to the backend to
|
||||||
|
/// put on screen, then clear the damage. A no-op when nothing is dirty. The frame counter
|
||||||
|
/// advances regardless, so callers can name frames.
|
||||||
|
fn present() void {
|
||||||
|
const dirty = damage.intersect(screenRect());
|
||||||
|
if (!dirty.isEmpty()) {
|
||||||
|
compositeInto(dirty);
|
||||||
|
backend.present(dirty);
|
||||||
|
}
|
||||||
|
damage = Rect.empty;
|
||||||
|
frames += 1;
|
||||||
|
|
||||||
|
// The first present after a native upgrade confirms the composited frame actually reached
|
||||||
|
// the shared scanout surface (the automated stand-in for "it's on screen").
|
||||||
|
if (pending_native_verify and !dirty.isEmpty()) {
|
||||||
|
pending_native_verify = false;
|
||||||
|
verifyNativePresent();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a pixel straight back from the shared scanout surface after a native present. The
|
||||||
|
/// surface starts zeroed, so a non-zero centre pixel means the compositor wrote the frame into
|
||||||
|
/// the pages the driver transfers-and-flushes from — that, plus the driver acking the present
|
||||||
|
/// over `.scanout`, is the serial proof the native path works.
|
||||||
|
fn verifyNativePresent() void {
|
||||||
|
const s = backend.surface();
|
||||||
|
const sample = s.pixels[@as(usize, s.height / 2) * s.stride + s.width / 2];
|
||||||
|
if (sample != 0) {
|
||||||
|
_ = system.write("display: native present verified\n");
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: native present FAILED (blank surface)\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A native scanout driver announced itself: map the shared surface it handed over, find its
|
||||||
|
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
||||||
|
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
||||||
|
/// unblocks the driver and it starts serving `.scanout`.
|
||||||
|
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
||||||
|
const cap = capability orelse return fail(reply);
|
||||||
|
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
||||||
|
const mapped = runtime.shm.map(cap) orelse return fail(reply);
|
||||||
|
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
|
||||||
|
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
||||||
|
// scanout. (The previous shared mapping leaks — there is no shm_unmap syscall yet — but the
|
||||||
|
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
||||||
|
const reattach = switch (backend) {
|
||||||
|
.virtio => true,
|
||||||
|
else => false,
|
||||||
|
};
|
||||||
|
|
||||||
|
backend = .{ .virtio = .{
|
||||||
|
.pixels = @ptrCast(@alignCast(mapped)),
|
||||||
|
.stride = stride,
|
||||||
|
.width = width,
|
||||||
|
.height = height,
|
||||||
|
.format = format,
|
||||||
|
.scanout = scanout,
|
||||||
|
} };
|
||||||
|
background = protocol.pack(format, 0x20, 0x30, 0x48); // re-pack the wallpaper for the mode
|
||||||
|
addDamage(screenRect()); // the whole new surface must be painted
|
||||||
|
pending_native_verify = true;
|
||||||
|
if (!reattach) pending_modeset_check = true; // the mode-set self-check runs once, on first upgrade
|
||||||
|
_ = system.timerOnce(service_endpoint, 50); // present once the driver is serving .scanout
|
||||||
|
_ = system.write(if (reattach)
|
||||||
|
"display: scanout re-attached\n"
|
||||||
|
else
|
||||||
|
"display: scanout upgraded to virtio-gpu\n");
|
||||||
|
return ok(reply);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// After the native upgrade is verified, prove the runtime-resolution-change and fenced-present
|
||||||
|
/// paths: query the driver's modes, switch to one that differs from the current, re-composite
|
||||||
|
/// the whole screen at the new size, and confirm the backend now reports that geometry. The
|
||||||
|
/// present goes through the driver's fenced flush, so a clean present is a vsync present.
|
||||||
|
fn modesetSelfCheck() void {
|
||||||
|
if (!backend.canModeSet()) return;
|
||||||
|
var mode_list: [4]backend_mod.Mode = undefined;
|
||||||
|
const count = backend.modes(&mode_list);
|
||||||
|
if (count == 0) {
|
||||||
|
_ = system.write("display: mode-set self-check: no modes reported\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const current = backend.info();
|
||||||
|
var target: ?backend_mod.Mode = null;
|
||||||
|
for (mode_list[0..count]) |m| {
|
||||||
|
if (m.width != current.width or m.height != current.height) {
|
||||||
|
target = m;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const wanted = target orelse {
|
||||||
|
_ = system.write("display: mode-set self-check: no alternate mode offered\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (!backend.setMode(wanted.width, wanted.height)) {
|
||||||
|
_ = system.write("display: mode set FAILED\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
addDamage(screenRect()); // repaint the whole screen at the new resolution, then present it
|
||||||
|
present();
|
||||||
|
|
||||||
|
const now = backend.info();
|
||||||
|
if (now.width == wanted.width and now.height == wanted.height) {
|
||||||
|
var line: [80]u8 = undefined;
|
||||||
|
_ = system.write(std.fmt.bufPrint(&line, "display: mode set to {d}x{d}, verified\n", .{ now.width, now.height }) catch "display: mode set, verified\n");
|
||||||
|
if (backend.hasVsync()) _ = system.write("display: vsync present ok\n");
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: mode set FAILED (geometry unchanged)\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- startup self-check -----------------------------------------------------
|
||||||
|
|
||||||
|
/// Prove the compositor wiring on the real backend: two overlapping opaque layers,
|
||||||
|
/// composited, must show the top layer in the overlap and the bottom layer outside it.
|
||||||
|
/// Exercises the whole path — mmap surfaces, the z-sort, damage, composite into the
|
||||||
|
/// backend surface — and reads the composited result back. Cleans up after itself.
|
||||||
|
fn selfCheck() void {
|
||||||
|
const format = backend.info().format;
|
||||||
|
const red = protocol.pack(format, 0xC0, 0x20, 0x20);
|
||||||
|
const green = protocol.pack(format, 0x20, 0xC0, 0x20);
|
||||||
|
const bottom = createLayer(100, 100, 80, 80, 0, true) orelse return fail_check("create");
|
||||||
|
const top = createLayer(140, 140, 80, 80, 1, true) orelse return fail_check("create");
|
||||||
|
_ = fillLayer(bottom, Rect.init(0, 0, 80, 80), red);
|
||||||
|
_ = fillLayer(top, Rect.init(0, 0, 80, 80), green);
|
||||||
|
present();
|
||||||
|
|
||||||
|
const surface = backend.surface();
|
||||||
|
const overlap = surface.pixels[@as(usize, 150) * surface.stride + 150]; // in both → top
|
||||||
|
const bottom_only = surface.pixels[@as(usize, 110) * surface.stride + 110]; // bottom only
|
||||||
|
|
||||||
|
_ = destroyLayer(top);
|
||||||
|
_ = destroyLayer(bottom);
|
||||||
|
present(); // repaint the self-check region back to the background
|
||||||
|
|
||||||
|
if (overlap == green and bottom_only == red) {
|
||||||
|
_ = system.write("display: compositor self-check ok\n");
|
||||||
|
} else {
|
||||||
|
_ = system.write("display: compositor self-check FAILED\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fail_check(_: []const u8) void {
|
||||||
|
_ = system.write("display: compositor self-check FAILED (setup)\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- service ----------------------------------------------------------------
|
||||||
|
|
||||||
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
|
service_endpoint = endpoint;
|
||||||
|
|
||||||
|
// Pick the scanout backend (GOP today). It logs the reason on failure.
|
||||||
|
backend = backend_mod.select() orelse return false;
|
||||||
|
const mode = backend.info();
|
||||||
|
background = protocol.pack(mode.format, 0x20, 0x30, 0x48); // a dark slate wallpaper
|
||||||
|
|
||||||
|
// Clear the whole screen through the compose surface → present path (double buffering:
|
||||||
|
// no direct-to-scanout drawing).
|
||||||
|
addDamage(screenRect());
|
||||||
|
present();
|
||||||
|
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = system.write(std.fmt.bufPrint(&line, "display: online {d}x{d} pitch {d} format {d}\n", .{
|
||||||
|
mode.width, mode.height, mode.pitch, mode.format,
|
||||||
|
}) catch "display: online\n");
|
||||||
|
_ = system.write("display: presented frame 0\n");
|
||||||
|
|
||||||
|
selfCheck();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeReply(reply: []u8, value: protocol.Reply) usize {
|
||||||
|
const bytes = std.mem.asBytes(&value);
|
||||||
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
|
return bytes.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn ok(reply: []u8) usize {
|
||||||
|
return writeReply(reply, .{ .status = 0 });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fail(reply: []u8) usize {
|
||||||
|
return writeReply(reply, .{ .status = -1 });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = sender;
|
||||||
|
if (message.len < protocol.request_size) return fail(reply);
|
||||||
|
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||||
|
const payload = message[protocol.request_size..];
|
||||||
|
// Switch on the raw operation value — an out-of-range one must fail cleanly, not
|
||||||
|
// panic an `@enumFromInt`.
|
||||||
|
switch (request.operation) {
|
||||||
|
@intFromEnum(protocol.Operation.info) => {
|
||||||
|
const m = backend.info();
|
||||||
|
return writeReply(reply, .{ .status = 0, .width = m.width, .height = m.height, .pitch = m.pitch, .format = m.format });
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.create_layer) => {
|
||||||
|
// x/y are signed coordinates carried in the u32 wire fields — reinterpret the
|
||||||
|
// bits (@bitCast), don't range-check (@intCast) which a negative would fail.
|
||||||
|
const slot = createLayer(@bitCast(request.x), @bitCast(request.y), request.width, request.height, request.z, request.visible != 0) orelse return fail(reply);
|
||||||
|
return writeReply(reply, .{ .status = 0, .layer = slot });
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.configure_layer) => {
|
||||||
|
return if (configureLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.z, request.visible != 0)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.destroy_layer) => {
|
||||||
|
return if (destroyLayer(request.layer)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.fill_rect) => {
|
||||||
|
const local = Rect.init(@bitCast(request.x), @bitCast(request.y), @intCast(request.width), @intCast(request.height));
|
||||||
|
return if (fillLayer(request.layer, local, request.colour)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.blit_tile) => {
|
||||||
|
return if (blitLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.width, request.height, payload)) ok(reply) else fail(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.damage) => {
|
||||||
|
const l = layerAt(request.layer) orelse return fail(reply);
|
||||||
|
const screen = Rect{ .x = l.x + @as(i32, @bitCast(request.x)), .y = l.y + @as(i32, @bitCast(request.y)), .w = @intCast(request.width), .h = @intCast(request.height) };
|
||||||
|
addDamage(screen.intersect(layerScreenRect(l)));
|
||||||
|
return ok(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.present) => {
|
||||||
|
present();
|
||||||
|
return ok(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.attach_scanout) => {
|
||||||
|
return attachScanout(request.x, request.width, request.height, request.colour, capability, reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.set_mode) => {
|
||||||
|
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
||||||
|
addDamage(screenRect()); // repaint the whole screen at the new resolution
|
||||||
|
present();
|
||||||
|
return ok(reply);
|
||||||
|
},
|
||||||
|
@intFromEnum(protocol.Operation.get_modes) => {
|
||||||
|
var list: [4]backend_mod.Mode = undefined;
|
||||||
|
const count = backend.modes(&list);
|
||||||
|
var response = protocol.ModesReply{ .status = 0, .count = @intCast(count), .modes = undefined };
|
||||||
|
for (0..protocol.max_modes) |i| {
|
||||||
|
response.modes[i] = if (i < count)
|
||||||
|
.{ .width = list[i].width, .height = list[i].height }
|
||||||
|
else
|
||||||
|
.{ .width = 0, .height = 0 };
|
||||||
|
}
|
||||||
|
const bytes = std.mem.asBytes(&response);
|
||||||
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
|
return bytes.len;
|
||||||
|
},
|
||||||
|
else => return fail(reply),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The only notification the compositor arms is the post-attach present timer: repaint the
|
||||||
|
/// screen into the freshly attached native surface, verify the frame landed, then run the
|
||||||
|
/// one-shot mode-set self-check (V5).
|
||||||
|
fn onNotification(badge: u64) void {
|
||||||
|
_ = badge;
|
||||||
|
present(); // native present + verify (first timer fire after the upgrade)
|
||||||
|
if (pending_modeset_check) {
|
||||||
|
pending_modeset_check = false;
|
||||||
|
modesetSelfCheck();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
.service = .display,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
.on_notification = onNotification,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,113 @@
|
|||||||
|
//! The display wire protocol — what a client says to the display service over its
|
||||||
|
//! well-known `.display` endpoint. extern-struct messages with an `Operation` tag, the
|
||||||
|
//! same shape as block/vfs/input protocols. The compositor owns the framebuffer and an
|
||||||
|
//! ordered stack of **layers**; a client creates layers, draws into them with these
|
||||||
|
//! operations, marks damage, and asks for a `present`. v1 surfaces are server-owned (a
|
||||||
|
//! client draws by command); shared-memory surfaces are a later milestone (docs/display.md).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub const Operation = enum(u32) {
|
||||||
|
/// info() -> { width, height, pitch, format }: the display's current mode.
|
||||||
|
info = 0,
|
||||||
|
/// create_layer(x, y, width, height, z) -> { layer }: a new server-owned surface.
|
||||||
|
create_layer = 1,
|
||||||
|
/// configure_layer(layer, x, y, z, visible): move, restack, show, or hide a layer.
|
||||||
|
configure_layer = 2,
|
||||||
|
/// destroy_layer(layer): release a layer.
|
||||||
|
destroy_layer = 3,
|
||||||
|
/// fill_rect(layer, x, y, width, height, colour): fill a rectangle of a layer.
|
||||||
|
fill_rect = 4,
|
||||||
|
/// blit_tile(layer, x, y, width, height, <inline pixels>): copy a small pixel tile in.
|
||||||
|
blit_tile = 5,
|
||||||
|
/// damage(layer, x, y, width, height): mark a region dirty for the next present.
|
||||||
|
damage = 6,
|
||||||
|
/// present(): composite the dirty layers and flush to the screen.
|
||||||
|
present = 7,
|
||||||
|
/// attach_scanout(x=stride, width, height, colour=format) + <surface capability>: a native
|
||||||
|
/// scanout driver announces itself, handing over the shared scanout surface as an `ipc_call`
|
||||||
|
/// send_cap. The compositor maps it, looks up the driver's `.scanout` present channel, and
|
||||||
|
/// upgrades off the GOP floor (docs/display-v2.md V4). `x` is the surface's row stride in
|
||||||
|
/// pixels, `colour` the DisplayFormat.
|
||||||
|
attach_scanout = 8,
|
||||||
|
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||||
|
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||||
|
set_mode = 9,
|
||||||
|
/// get_modes() -> ModesReply: the resolutions the display can switch to (empty on GOP).
|
||||||
|
get_modes = 10,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The fixed request header. A `blit_tile`'s pixel payload (width*height 32-bit pixels)
|
||||||
|
/// follows this header inline in the same message, up to `maximum_payload`.
|
||||||
|
pub const Request = extern struct {
|
||||||
|
operation: u32,
|
||||||
|
layer: u32 = 0, // create/configure/destroy/fill/blit/damage: the target layer
|
||||||
|
x: u32 = 0,
|
||||||
|
y: u32 = 0,
|
||||||
|
width: u32 = 0,
|
||||||
|
height: u32 = 0,
|
||||||
|
z: u32 = 0, // create_layer / configure_layer: stacking order (higher = in front)
|
||||||
|
colour: u32 = 0, // fill_rect: the fill colour (native pixel value)
|
||||||
|
visible: u32 = 1, // configure_layer: 0 hides the layer
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Reply = extern struct {
|
||||||
|
status: i32, // 0 on success, negative on failure
|
||||||
|
reserved: u32 = 0,
|
||||||
|
// info():
|
||||||
|
width: u32 = 0,
|
||||||
|
height: u32 = 0,
|
||||||
|
pitch: u32 = 0,
|
||||||
|
format: u32 = 0, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||||
|
// create_layer():
|
||||||
|
layer: u32 = 0,
|
||||||
|
reserved2: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One selectable display mode.
|
||||||
|
pub const Mode = extern struct { width: u32, height: u32 };
|
||||||
|
pub const max_modes = 4;
|
||||||
|
|
||||||
|
/// The reply to `get_modes`: a small fixed list of resolutions the display can switch to.
|
||||||
|
pub const ModesReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
count: u32,
|
||||||
|
modes: [max_modes]Mode,
|
||||||
|
};
|
||||||
|
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||||
|
|
||||||
|
/// The IPC message size — the kernel caps every message at `MESSAGE_MAXIMUM` (256 bytes,
|
||||||
|
/// system/kernel/ipc-synchronous.zig), so this matches it (a larger receive/reply buffer
|
||||||
|
/// is rejected with -E2BIG). A `blit_tile` therefore carries only a *small* tile inline —
|
||||||
|
/// `maximum_payload` bytes = up to 54 pixels, enough for a cursor or small sprite; larger
|
||||||
|
/// bitmaps are the deferred shared-memory surface path (docs/display.md).
|
||||||
|
pub const message_maximum: usize = 256;
|
||||||
|
pub const request_size: usize = @sizeOf(Request);
|
||||||
|
pub const reply_size: usize = @sizeOf(Reply);
|
||||||
|
pub const maximum_payload: usize = message_maximum - request_size;
|
||||||
|
|
||||||
|
/// Pack an 8-bit-per-channel colour into the display's native 32-bit pixel for `format`
|
||||||
|
/// (a device-abi `DisplayFormat`: 0 = rgbx, 1 = bgrx). Shared so a `colour` in a
|
||||||
|
/// `fill_rect` request means the same thing to the client that sends it and the
|
||||||
|
/// compositor that paints it. Little-endian memory, reserved byte 0: rgbx puts red in
|
||||||
|
/// the low byte, bgrx puts blue there.
|
||||||
|
pub fn pack(format: u32, r: u8, g: u8, b: u8) u32 {
|
||||||
|
const rr: u32 = r;
|
||||||
|
const gg: u32 = g;
|
||||||
|
const bb: u32 = b;
|
||||||
|
return switch (format) {
|
||||||
|
1 => bb | (gg << 8) | (rr << 16), // bgrx
|
||||||
|
else => rr | (gg << 8) | (bb << 16), // rgbx
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "pack encodes native byte order for rgbx and bgrx" {
|
||||||
|
// rgbx: red in the low byte, blue in byte 2.
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0000_00AA), pack(0, 0xAA, 0, 0));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(0, 0, 0, 0xAA));
|
||||||
|
// bgrx: blue in the low byte, red in byte 2.
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0000_00AA), pack(1, 0, 0, 0xAA));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(1, 0xAA, 0, 0));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x0000_3020), pack(0, 0x20, 0x30, 0)); // green in byte 1
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
//! The scanout wire protocol — what the compositor says to a native scanout driver (e.g.
|
||||||
|
//! virtio-gpu) over its well-known `.scanout` endpoint to put a composited frame on screen.
|
||||||
|
//! The driver owns the panel and the shared scanout surface it handed the compositor (via the
|
||||||
|
//! display service's `attach_scanout`); the compositor composites into that surface, then asks
|
||||||
|
//! the driver to present a damaged rectangle. Tiny by design — one present request. Separate
|
||||||
|
//! from the display protocol because the directions differ: clients call the compositor over
|
||||||
|
//! `.display`; the compositor calls the driver over `.scanout`. See docs/display-v2.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub const Operation = enum(u32) {
|
||||||
|
/// present(x, y, width, height): put the given rectangle of the shared scanout surface on
|
||||||
|
/// the panel (on virtio-gpu: transfer-to-host of the region, then a fenced resource flush).
|
||||||
|
present = 0,
|
||||||
|
/// get_modes() -> ModesReply: the display modes this scanout can switch to (V5).
|
||||||
|
get_modes = 1,
|
||||||
|
/// set_mode(width, height): change the scanout resolution — the shared surface is sized to
|
||||||
|
/// the largest mode, so this just re-points the scanout rectangle; the surface is unchanged.
|
||||||
|
set_mode = 2,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Request = extern struct {
|
||||||
|
operation: u32,
|
||||||
|
x: u32 = 0,
|
||||||
|
y: u32 = 0,
|
||||||
|
width: u32 = 0,
|
||||||
|
height: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Reply = extern struct {
|
||||||
|
status: i32, // 0 on success, negative on failure
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One offered display mode.
|
||||||
|
pub const Mode = extern struct { width: u32, height: u32 };
|
||||||
|
pub const max_modes = 4;
|
||||||
|
|
||||||
|
/// The reply to `get_modes`: a small fixed list of modes.
|
||||||
|
pub const ModesReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
count: u32,
|
||||||
|
modes: [max_modes]Mode,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const message_maximum: usize = 64;
|
||||||
|
pub const request_size: usize = @sizeOf(Request);
|
||||||
|
pub const reply_size: usize = @sizeOf(Reply);
|
||||||
|
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,108 @@
|
|||||||
|
//! system/services/fat/fat-test — a client that proves the FAT mount end to end:
|
||||||
|
//! it waits for the fat server to mount the USB volume at /mnt/usb, lists the
|
||||||
|
//! root directory through the VFS (which routes /mnt/usb to the fat backend), and
|
||||||
|
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
||||||
|
//! kernel test spawns it alongside init.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const fs = runtime.fs;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
_ = init;
|
||||||
|
|
||||||
|
// Wait for /mnt/usb to be mounted — the fat server races us at boot (it must
|
||||||
|
// bring up the whole USB storage chain first).
|
||||||
|
var opened: ?fs.Directory = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (opened == null and tries < 1400) : (tries += 1) {
|
||||||
|
opened = fs.openDirectory("/mnt/usb");
|
||||||
|
if (opened == null) runtime.system.sleep(50);
|
||||||
|
}
|
||||||
|
var dir = opened orelse {
|
||||||
|
_ = runtime.system.write("fat-test: /mnt/usb never became available\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var count: u32 = 0;
|
||||||
|
var entry: fs.Entry = .{};
|
||||||
|
while (dir.next(&entry)) {
|
||||||
|
writeLine("fat-test: entry '{s}' kind={d} size={d}\n", .{ entry.name(), @intFromEnum(entry.kind), entry.size });
|
||||||
|
count += 1;
|
||||||
|
if (count > 32) break;
|
||||||
|
}
|
||||||
|
dir.close();
|
||||||
|
writeLine("fat-test: listed {d} entries\n", .{count});
|
||||||
|
|
||||||
|
// Read a known file off the boot volume through the mount (best effort): the
|
||||||
|
// kernel image is an ELF, so its first bytes are the ELF magic.
|
||||||
|
if (fs.open("/mnt/usb/system/kernel", .{})) |opened_file| {
|
||||||
|
var file = opened_file;
|
||||||
|
var magic: [4]u8 = undefined;
|
||||||
|
const n = file.read(&magic) orelse 0;
|
||||||
|
file.close();
|
||||||
|
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
||||||
|
_ = runtime.system.write("fat-test: read /mnt/usb/system/kernel ELF magic ok\n");
|
||||||
|
} else {
|
||||||
|
writeLine("fat-test: /mnt/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Exercise directory + file mutation through the mount: mkdir, create a file
|
||||||
|
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
||||||
|
if (fs.makeDirectory("/mnt/usb/TESTDIR")) {
|
||||||
|
var wrote = false;
|
||||||
|
if (fs.open("/mnt/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||||
|
var f = created;
|
||||||
|
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
||||||
|
f.close();
|
||||||
|
}
|
||||||
|
// The created file carries a real modification time (stamped from the RTC).
|
||||||
|
var mtime_ok = false;
|
||||||
|
if (fs.attributes("/mnt/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
||||||
|
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
||||||
|
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
||||||
|
}
|
||||||
|
if (mtime_ok) _ = runtime.system.write("fat-test: mtime ok\n");
|
||||||
|
|
||||||
|
// Rename it, then read from the new name and confirm the old name is gone.
|
||||||
|
const renamed = fs.rename("/mnt/usb/TESTDIR/HELLO.TXT", "/mnt/usb/TESTDIR/RENAMED.TXT");
|
||||||
|
const old_gone = !fs.exists("/mnt/usb/TESTDIR/HELLO.TXT");
|
||||||
|
if (renamed and old_gone) _ = runtime.system.write("fat-test: rename ok\n");
|
||||||
|
var readback = false;
|
||||||
|
if (fs.open("/mnt/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||||
|
var f = reopened;
|
||||||
|
var buf: [16]u8 = undefined;
|
||||||
|
const got = f.read(&buf) orelse 0;
|
||||||
|
f.close();
|
||||||
|
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
||||||
|
}
|
||||||
|
const removed = fs.remove("/mnt/usb/TESTDIR/RENAMED.TXT");
|
||||||
|
const gone = !fs.exists("/mnt/usb/TESTDIR/RENAMED.TXT");
|
||||||
|
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
||||||
|
_ = runtime.system.write("fat-test: mutations ok\n");
|
||||||
|
} else {
|
||||||
|
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
_ = runtime.system.write("fat-test: mkdir /mnt/usb/TESTDIR failed\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
if (count > 0) {
|
||||||
|
while (true) {
|
||||||
|
_ = runtime.system.write("fat-test: ok\n");
|
||||||
|
runtime.system.sleep(1000);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = runtime.system.write("fat-test: root listing was empty\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,249 @@
|
|||||||
|
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||||
|
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||||
|
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||||
|
//! the VFS at /mnt/usb. From then on the VFS forwards every open/read/write/
|
||||||
|
//! status/readdir/close under /mnt/usb to this server, which serves the same
|
||||||
|
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||||
|
//!
|
||||||
|
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||||
|
//! block driver by physical address, and the engine copies sectors in and out of
|
||||||
|
//! it.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const engine = @import("engine.zig");
|
||||||
|
const on_disk = @import("on-disk.zig");
|
||||||
|
const protocol = runtime.vfs_protocol;
|
||||||
|
const dma = runtime.dma;
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
const mount_point = "/mnt/usb";
|
||||||
|
|
||||||
|
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||||
|
// buffer the driver reads/writes by physical address.
|
||||||
|
const IpcBlock = struct {
|
||||||
|
device: runtime.block.Device,
|
||||||
|
bounce: dma.Region,
|
||||||
|
|
||||||
|
fn readBlock(context: *anyopaque, lba: u64, buffer: []u8) bool {
|
||||||
|
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||||
|
if (!self.device.read(lba, 1, self.bounce.physical)) return false;
|
||||||
|
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||||
|
@memcpy(buffer[0..512], source[0..512]);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
fn writeBlock(context: *anyopaque, lba: u64, buffer: []const u8) bool {
|
||||||
|
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||||
|
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||||
|
@memcpy(destination[0..512], buffer[0..512]);
|
||||||
|
if (!self.device.write(lba, 1, self.bounce.physical)) return false;
|
||||||
|
device_dirty = true; // a block reached the device; a close will flush it
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
var ipc_block: IpcBlock = undefined;
|
||||||
|
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||||
|
// file close — so writes are committed to stable media before a power-off.
|
||||||
|
var device_dirty: bool = false;
|
||||||
|
var filesystem: engine.FileSystem = undefined;
|
||||||
|
|
||||||
|
// Open handles the VFS holds against this backend: each maps a node id to a
|
||||||
|
// resolved engine node.
|
||||||
|
const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 };
|
||||||
|
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||||
|
|
||||||
|
fn allocOpen() ?usize {
|
||||||
|
for (&open_nodes, 0..) |*o, i| {
|
||||||
|
if (!o.used) return i;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn openAt(id: u64) ?*OpenNode {
|
||||||
|
if (id >= open_nodes.len) return null;
|
||||||
|
const o = &open_nodes[@intCast(id)];
|
||||||
|
return if (o.used) o else null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
|
||||||
|
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
|
||||||
|
const n = @min(payload.len, out.len - protocol.reply_size);
|
||||||
|
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
|
||||||
|
return protocol.reply_size + n;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fail(out: []u8) usize {
|
||||||
|
return writeReply(out, .{ .status = -1 }, &.{});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||||
|
_ = runtime.system.write("/system/services/fat: starting, waiting for a block device\n");
|
||||||
|
const device = runtime.block.open() orelse {
|
||||||
|
_ = runtime.system.write("/system/services/fat: no block device (no storage attached)\n");
|
||||||
|
return false; // clean exit: nothing to serve
|
||||||
|
};
|
||||||
|
const geometry = device.geometry() orelse {
|
||||||
|
_ = runtime.system.write("/system/services/fat: block geometry unavailable\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
ipc_block = .{ .device = device, .bounce = dma.alloc(4096, dma.coherent) orelse return false };
|
||||||
|
|
||||||
|
const block_device = engine.BlockDevice{
|
||||||
|
.context = &ipc_block,
|
||||||
|
.block_size = geometry.block_size,
|
||||||
|
.block_count = geometry.block_count,
|
||||||
|
.readBlockFn = IpcBlock.readBlock,
|
||||||
|
.writeBlockFn = IpcBlock.writeBlock,
|
||||||
|
};
|
||||||
|
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||||
|
_ = runtime.system.write("/system/services/fat: not a FAT filesystem\n");
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
writeLine("/system/services/fat: mounted FAT ({s}, {d} clusters, partition lba {d})\n", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||||
|
|
||||||
|
// Mount ourselves into the VFS namespace at /mnt/usb (retry while the VFS
|
||||||
|
// comes up). From here the VFS routes /mnt/usb/... to this server.
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (tries < 100) : (tries += 1) {
|
||||||
|
if (runtime.fs.mount(mount_point, endpoint)) {
|
||||||
|
writeLine("/system/services/fat: mounted {s}\n", .{mount_point});
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
runtime.system.sleep(50);
|
||||||
|
}
|
||||||
|
_ = runtime.system.write("/system/services/fat: could not mount into the VFS\n");
|
||||||
|
return true; // still serve directly, even if the namespace mount didn't take
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||||
|
|
||||||
|
// Split a path into its parent directory and final component: "/a/b" -> ("/a",
|
||||||
|
// "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||||
|
fn splitParent(path: []const u8) ParentLeaf {
|
||||||
|
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||||
|
return .{
|
||||||
|
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||||
|
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
|
||||||
|
var node = filesystem.resolve(path);
|
||||||
|
if (node == null and flags & protocol.create != 0) {
|
||||||
|
const split = splitParent(path);
|
||||||
|
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||||
|
node = filesystem.createFile(parent, split.leaf);
|
||||||
|
}
|
||||||
|
var resolved = node orelse return fail(out);
|
||||||
|
// O_TRUNC: replace an existing file's contents rather than overwriting in place
|
||||||
|
// (frees the old chain, so a shorter rewrite leaves no stale tail).
|
||||||
|
if (flags & protocol.truncate != 0 and !resolved.is_directory) {
|
||||||
|
filesystem.truncate(&resolved);
|
||||||
|
}
|
||||||
|
const index = allocOpen() orelse return fail(out);
|
||||||
|
open_nodes[index] = .{ .used = true, .node = resolved };
|
||||||
|
return writeReply(out, .{ .status = 0, .node = index }, &.{});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||||
|
_ = capability;
|
||||||
|
_ = sender;
|
||||||
|
if (message.len < protocol.request_size) return fail(out);
|
||||||
|
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||||
|
const payload = message[protocol.request_size..];
|
||||||
|
|
||||||
|
// Stamp create/write with the current wall-clock time (mtime). Cheap, and it
|
||||||
|
// keeps the engine pure (it takes the time as data, not a syscall).
|
||||||
|
filesystem.current_time_epoch = runtime.system.wallClock();
|
||||||
|
|
||||||
|
switch (request.operation) {
|
||||||
|
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags),
|
||||||
|
.read => {
|
||||||
|
const o = openAt(request.node) orelse return fail(out);
|
||||||
|
var buffer: [protocol.maximum_payload]u8 = undefined;
|
||||||
|
const want = @min(@as(usize, request.len), buffer.len);
|
||||||
|
const n = filesystem.readFile(o.node, @intCast(request.offset), buffer[0..want]);
|
||||||
|
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, buffer[0..n]);
|
||||||
|
},
|
||||||
|
.write => {
|
||||||
|
const o = openAt(request.node) orelse return fail(out);
|
||||||
|
const data = payload[0..@min(payload.len, request.len)];
|
||||||
|
const n = filesystem.writeFile(&o.node, @intCast(request.offset), data);
|
||||||
|
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
|
||||||
|
},
|
||||||
|
.status => {
|
||||||
|
const o = openAt(request.node) orelse return fail(out);
|
||||||
|
const kind: protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||||
|
const status = protocol.FileStatus{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime };
|
||||||
|
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&status));
|
||||||
|
},
|
||||||
|
.readdir => {
|
||||||
|
const o = openAt(request.node) orelse return fail(out);
|
||||||
|
if (!o.node.is_directory) return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
|
||||||
|
const listing = filesystem.listEntry(o.node, @intCast(request.offset)) orelse return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
|
||||||
|
const kind: protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||||
|
const header = protocol.DirectoryEntry{ .kind = @intFromEnum(kind), .name_len = @intCast(listing.name_len), .size = listing.size };
|
||||||
|
var buffer: [protocol.maximum_payload]u8 = undefined;
|
||||||
|
@memcpy(buffer[0..protocol.directory_entry_size], std.mem.asBytes(&header));
|
||||||
|
const nlen = @min(listing.name_len, buffer.len - protocol.directory_entry_size);
|
||||||
|
@memcpy(buffer[protocol.directory_entry_size..][0..nlen], listing.name_buffer[0..nlen]);
|
||||||
|
const total = protocol.directory_entry_size + nlen;
|
||||||
|
return writeReply(out, .{ .status = 0, .len = @intCast(total) }, buffer[0..total]);
|
||||||
|
},
|
||||||
|
.close => {
|
||||||
|
if (openAt(request.node)) |o| o.used = false;
|
||||||
|
// Durable-on-close: if any block reached the device since the last
|
||||||
|
// flush, commit its cache to stable media now (best-effort). This is
|
||||||
|
// what makes init's shutdown log flush survive a real power-off, and is
|
||||||
|
// the right default for removable media the user may unplug.
|
||||||
|
if (device_dirty) {
|
||||||
|
_ = ipc_block.device.flush();
|
||||||
|
device_dirty = false;
|
||||||
|
}
|
||||||
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
|
},
|
||||||
|
.mkdir => {
|
||||||
|
const split = splitParent(payload[0..@min(payload.len, request.len)]);
|
||||||
|
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||||
|
if (filesystem.createDirectory(parent, split.leaf) == null) return fail(out);
|
||||||
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
|
},
|
||||||
|
.unlink => {
|
||||||
|
const split = splitParent(payload[0..@min(payload.len, request.len)]);
|
||||||
|
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||||
|
if (!filesystem.removeFile(parent, split.leaf)) return fail(out);
|
||||||
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
|
},
|
||||||
|
.rename => {
|
||||||
|
const both = payload[0..@min(payload.len, request.len)];
|
||||||
|
const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out);
|
||||||
|
const old_split = splitParent(both[0..sep]);
|
||||||
|
const new_split = splitParent(both[sep + 1 ..]);
|
||||||
|
// Same-directory rename only.
|
||||||
|
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return fail(out);
|
||||||
|
const parent = filesystem.resolve(old_split.parent) orelse return fail(out);
|
||||||
|
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return fail(out);
|
||||||
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
|
},
|
||||||
|
// A backend is never itself a mount target.
|
||||||
|
.mount, .unmount => return fail(out),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
runtime.service.run(protocol.message_maximum, .{
|
||||||
|
.service = .fat,
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,310 @@
|
|||||||
|
//! The on-disk layout of a FAT filesystem — the boot sector / BIOS Parameter
|
||||||
|
//! Block, directory entries, long-file-name entries, and the FAT32 FSInfo — as
|
||||||
|
//! `align(1)` extern structs that bit-cast straight out of a 512-byte sector
|
||||||
|
//! (multi-byte fields are little-endian, like usb-abi.zig). Pure data, plus the
|
||||||
|
//! cluster-count FAT-type detection. Host-testable.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// The BIOS Parameter Block, common to FAT12/16/32 (offset 0..36 of the boot
|
||||||
|
/// sector). The extended part that follows differs by FAT type.
|
||||||
|
pub const BiosParameterBlock = extern struct {
|
||||||
|
jump: [3]u8,
|
||||||
|
oem_name: [8]u8,
|
||||||
|
bytes_per_sector: u16 align(1),
|
||||||
|
sectors_per_cluster: u8,
|
||||||
|
reserved_sector_count: u16 align(1),
|
||||||
|
fat_count: u8,
|
||||||
|
root_entry_count: u16 align(1),
|
||||||
|
total_sectors_16: u16 align(1),
|
||||||
|
media: u8,
|
||||||
|
fat_size_16: u16 align(1),
|
||||||
|
sectors_per_track: u16 align(1),
|
||||||
|
head_count: u16 align(1),
|
||||||
|
hidden_sectors: u32 align(1),
|
||||||
|
total_sectors_32: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The FAT12/16 extended boot record (offset 36).
|
||||||
|
pub const ExtendedBootRecord16 = extern struct {
|
||||||
|
drive_number: u8,
|
||||||
|
reserved: u8,
|
||||||
|
boot_signature: u8,
|
||||||
|
volume_id: u32 align(1),
|
||||||
|
volume_label: [11]u8,
|
||||||
|
filesystem_type: [8]u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The FAT32 extended boot record (offset 36).
|
||||||
|
pub const ExtendedBootRecord32 = extern struct {
|
||||||
|
fat_size_32: u32 align(1),
|
||||||
|
extended_flags: u16 align(1),
|
||||||
|
filesystem_version: u16 align(1),
|
||||||
|
root_cluster: u32 align(1),
|
||||||
|
filesystem_information_sector: u16 align(1),
|
||||||
|
backup_boot_sector: u16 align(1),
|
||||||
|
reserved: [12]u8,
|
||||||
|
drive_number: u8,
|
||||||
|
reserved1: u8,
|
||||||
|
boot_signature: u8,
|
||||||
|
volume_id: u32 align(1),
|
||||||
|
volume_label: [11]u8,
|
||||||
|
filesystem_type: [8]u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A 32-byte directory entry (8.3 short name form).
|
||||||
|
pub const DirectoryEntry = extern struct {
|
||||||
|
name: [11]u8, // 8 name + 3 extension, space-padded
|
||||||
|
attributes: u8,
|
||||||
|
reserved_nt: u8,
|
||||||
|
creation_time_tenth: u8,
|
||||||
|
creation_time: u16 align(1),
|
||||||
|
creation_date: u16 align(1),
|
||||||
|
last_access_date: u16 align(1),
|
||||||
|
first_cluster_high: u16 align(1),
|
||||||
|
write_time: u16 align(1),
|
||||||
|
write_date: u16 align(1),
|
||||||
|
first_cluster_low: u16 align(1),
|
||||||
|
file_size: u32 align(1),
|
||||||
|
|
||||||
|
pub fn firstCluster(self: DirectoryEntry) u32 {
|
||||||
|
return (@as(u32, self.first_cluster_high) << 16) | self.first_cluster_low;
|
||||||
|
}
|
||||||
|
pub fn setFirstCluster(self: *DirectoryEntry, cluster: u32) void {
|
||||||
|
self.first_cluster_low = @truncate(cluster);
|
||||||
|
self.first_cluster_high = @truncate(cluster >> 16);
|
||||||
|
}
|
||||||
|
pub fn isFree(self: DirectoryEntry) bool {
|
||||||
|
return self.name[0] == 0x00 or self.name[0] == 0xE5;
|
||||||
|
}
|
||||||
|
pub fn isEnd(self: DirectoryEntry) bool {
|
||||||
|
return self.name[0] == 0x00;
|
||||||
|
}
|
||||||
|
pub fn isDirectory(self: DirectoryEntry) bool {
|
||||||
|
return self.attributes & attribute_directory != 0;
|
||||||
|
}
|
||||||
|
pub fn isLongName(self: DirectoryEntry) bool {
|
||||||
|
return self.attributes & attribute_long_name_mask == attribute_long_name;
|
||||||
|
}
|
||||||
|
pub fn isVolumeLabel(self: DirectoryEntry) bool {
|
||||||
|
return self.attributes & attribute_volume_id != 0 and !self.isLongName();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A 32-byte long-file-name entry (attributes == 0x0F). A sequence of these
|
||||||
|
/// precedes the 8.3 entry they name, each carrying 13 UTF-16 code units.
|
||||||
|
pub const LongNameEntry = extern struct {
|
||||||
|
order: u8,
|
||||||
|
name1: [5]u16 align(1),
|
||||||
|
attributes: u8,
|
||||||
|
kind: u8,
|
||||||
|
checksum: u8,
|
||||||
|
name2: [6]u16 align(1),
|
||||||
|
first_cluster_low: u16 align(1),
|
||||||
|
name3: [2]u16 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The FAT32 FSInfo sector (usually sector 1): advisory free-cluster bookkeeping.
|
||||||
|
pub const FileSystemInformation = extern struct {
|
||||||
|
lead_signature: u32 align(1), // 0x41615252
|
||||||
|
reserved1: [480]u8,
|
||||||
|
struct_signature: u32 align(1), // 0x61417272
|
||||||
|
free_count: u32 align(1),
|
||||||
|
next_free: u32 align(1),
|
||||||
|
reserved2: [12]u8,
|
||||||
|
trail_signature: u32 align(1), // 0xAA550000
|
||||||
|
};
|
||||||
|
|
||||||
|
// Directory-entry attribute bits.
|
||||||
|
pub const attribute_read_only: u8 = 0x01;
|
||||||
|
pub const attribute_hidden: u8 = 0x02;
|
||||||
|
pub const attribute_system: u8 = 0x04;
|
||||||
|
pub const attribute_volume_id: u8 = 0x08;
|
||||||
|
pub const attribute_directory: u8 = 0x10;
|
||||||
|
pub const attribute_archive: u8 = 0x20;
|
||||||
|
pub const attribute_long_name: u8 = 0x0F; // read_only|hidden|system|volume_id
|
||||||
|
pub const attribute_long_name_mask: u8 = 0x3F;
|
||||||
|
|
||||||
|
// FSInfo signatures.
|
||||||
|
pub const fsinfo_lead_signature: u32 = 0x41615252;
|
||||||
|
pub const fsinfo_struct_signature: u32 = 0x61417272;
|
||||||
|
pub const fsinfo_trail_signature: u32 = 0xAA550000;
|
||||||
|
|
||||||
|
/// End-of-chain markers (a cluster value >= these ends a chain).
|
||||||
|
pub const end_of_chain_12: u32 = 0xFF8;
|
||||||
|
pub const end_of_chain_16: u32 = 0xFFF8;
|
||||||
|
pub const end_of_chain_32: u32 = 0x0FFFFFF8;
|
||||||
|
pub const bad_cluster_32: u32 = 0x0FFFFFF7;
|
||||||
|
|
||||||
|
pub const free_cluster: u32 = 0;
|
||||||
|
pub const boot_signature_offset: usize = 510; // 0x55 0xAA at the end of the boot sector
|
||||||
|
|
||||||
|
pub const FatType = enum { fat12, fat16, fat32 };
|
||||||
|
|
||||||
|
/// The geometry derived from the BPB, plus the FAT type (by the Microsoft
|
||||||
|
/// cluster-count rule: <4085 FAT12, <65525 FAT16, else FAT32).
|
||||||
|
pub const Geometry = struct {
|
||||||
|
fat_type: FatType,
|
||||||
|
bytes_per_sector: u32,
|
||||||
|
sectors_per_cluster: u32,
|
||||||
|
reserved_sector_count: u32,
|
||||||
|
fat_count: u32,
|
||||||
|
fat_size_sectors: u32, // per FAT
|
||||||
|
root_entry_count: u32, // FAT12/16
|
||||||
|
root_cluster: u32, // FAT32
|
||||||
|
first_data_sector: u32,
|
||||||
|
total_sectors: u32,
|
||||||
|
cluster_count: u32,
|
||||||
|
fsinfo_sector: u32, // FAT32
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Derive the geometry (and FAT type) from a boot sector's first 512 bytes.
|
||||||
|
/// Returns null if the sector is not a plausible FAT boot sector.
|
||||||
|
pub fn geometryOf(sector: []const u8) ?Geometry {
|
||||||
|
if (sector.len < 512) return null;
|
||||||
|
if (sector[boot_signature_offset] != 0x55 or sector[boot_signature_offset + 1] != 0xAA) return null;
|
||||||
|
const bpb = std.mem.bytesToValue(BiosParameterBlock, sector[0..@sizeOf(BiosParameterBlock)]);
|
||||||
|
if (bpb.bytes_per_sector == 0 or bpb.sectors_per_cluster == 0 or bpb.fat_count == 0) return null;
|
||||||
|
|
||||||
|
const fat_size_16: u32 = bpb.fat_size_16;
|
||||||
|
var fat_size: u32 = fat_size_16;
|
||||||
|
var root_cluster: u32 = 0;
|
||||||
|
var fsinfo_sector: u32 = 0;
|
||||||
|
if (fat_size_16 == 0) {
|
||||||
|
const ebr = std.mem.bytesToValue(ExtendedBootRecord32, sector[36 .. 36 + @sizeOf(ExtendedBootRecord32)]);
|
||||||
|
fat_size = ebr.fat_size_32;
|
||||||
|
root_cluster = ebr.root_cluster;
|
||||||
|
fsinfo_sector = ebr.filesystem_information_sector;
|
||||||
|
}
|
||||||
|
|
||||||
|
const total_sectors: u32 = if (bpb.total_sectors_16 != 0) bpb.total_sectors_16 else bpb.total_sectors_32;
|
||||||
|
const root_dir_sectors = (@as(u32, bpb.root_entry_count) * 32 + bpb.bytes_per_sector - 1) / bpb.bytes_per_sector;
|
||||||
|
const first_data_sector = bpb.reserved_sector_count + bpb.fat_count * fat_size + root_dir_sectors;
|
||||||
|
if (total_sectors < first_data_sector) return null;
|
||||||
|
const data_sectors = total_sectors - first_data_sector;
|
||||||
|
const cluster_count = data_sectors / bpb.sectors_per_cluster;
|
||||||
|
|
||||||
|
const fat_type: FatType = if (cluster_count < 4085) .fat12 else if (cluster_count < 65525) .fat16 else .fat32;
|
||||||
|
|
||||||
|
return .{
|
||||||
|
.fat_type = fat_type,
|
||||||
|
.bytes_per_sector = bpb.bytes_per_sector,
|
||||||
|
.sectors_per_cluster = bpb.sectors_per_cluster,
|
||||||
|
.reserved_sector_count = bpb.reserved_sector_count,
|
||||||
|
.fat_count = bpb.fat_count,
|
||||||
|
.fat_size_sectors = fat_size,
|
||||||
|
.root_entry_count = bpb.root_entry_count,
|
||||||
|
.root_cluster = root_cluster,
|
||||||
|
.first_data_sector = first_data_sector,
|
||||||
|
.total_sectors = total_sectors,
|
||||||
|
.cluster_count = cluster_count,
|
||||||
|
.fsinfo_sector = fsinfo_sector,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- DOS date/time <-> Unix epoch --------------------------------------------
|
||||||
|
//
|
||||||
|
// FAT stamps a file's modification time as two 16-bit DOS fields. There is no
|
||||||
|
// timezone, so danos treats them as UTC. `date`: year-1980(7)|month(4)|day(5);
|
||||||
|
// `time`: hour(5)|minute(6)|(second/2)(5).
|
||||||
|
|
||||||
|
fn isLeapYear(year: u32) bool {
|
||||||
|
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
const days_in_month = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||||
|
|
||||||
|
/// Convert a FAT date+time to Unix epoch seconds (UTC). Returns 0 for an unset
|
||||||
|
/// (zero) date.
|
||||||
|
pub fn fatToEpoch(date: u16, time: u16) u64 {
|
||||||
|
if (date == 0) return 0;
|
||||||
|
const day: u32 = date & 0x1F;
|
||||||
|
const month: u32 = (date >> 5) & 0x0F;
|
||||||
|
const year: u32 = 1980 + (date >> 9);
|
||||||
|
if (month < 1 or month > 12 or day < 1) return 0;
|
||||||
|
const second: u32 = @as(u32, time & 0x1F) * 2;
|
||||||
|
const minute: u32 = (time >> 5) & 0x3F;
|
||||||
|
const hour: u32 = (time >> 11) & 0x1F;
|
||||||
|
|
||||||
|
var days: u64 = 0;
|
||||||
|
var y: u32 = 1970;
|
||||||
|
while (y < year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||||
|
var m: u32 = 1;
|
||||||
|
while (m < month) : (m += 1) {
|
||||||
|
days += days_in_month[m - 1];
|
||||||
|
if (m == 2 and isLeapYear(year)) days += 1;
|
||||||
|
}
|
||||||
|
days += day - 1;
|
||||||
|
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const FatDateTime = struct { date: u16, time: u16 };
|
||||||
|
|
||||||
|
/// Convert Unix epoch seconds (UTC) to a FAT date+time. Returns {0,0} for epoch 0 or
|
||||||
|
/// any time before 1980 (which DOS cannot represent).
|
||||||
|
pub fn epochToFatDateTime(epoch: u64) FatDateTime {
|
||||||
|
if (epoch == 0) return .{ .date = 0, .time = 0 };
|
||||||
|
var remaining = epoch;
|
||||||
|
const second: u32 = @intCast(remaining % 60);
|
||||||
|
remaining /= 60;
|
||||||
|
const minute: u32 = @intCast(remaining % 60);
|
||||||
|
remaining /= 60;
|
||||||
|
const hour: u32 = @intCast(remaining % 24);
|
||||||
|
remaining /= 24;
|
||||||
|
var days: u32 = @intCast(remaining); // whole days since 1970-01-01
|
||||||
|
|
||||||
|
var year: u32 = 1970;
|
||||||
|
while (true) {
|
||||||
|
const y_days: u32 = if (isLeapYear(year)) 366 else 365;
|
||||||
|
if (days < y_days) break;
|
||||||
|
days -= y_days;
|
||||||
|
year += 1;
|
||||||
|
}
|
||||||
|
if (year < 1980) return .{ .date = 0, .time = 0 };
|
||||||
|
var month: u32 = 1;
|
||||||
|
while (true) {
|
||||||
|
var m_days: u32 = days_in_month[month - 1];
|
||||||
|
if (month == 2 and isLeapYear(year)) m_days += 1;
|
||||||
|
if (days < m_days) break;
|
||||||
|
days -= m_days;
|
||||||
|
month += 1;
|
||||||
|
}
|
||||||
|
const day = days + 1;
|
||||||
|
return .{
|
||||||
|
.date = @intCast(((year - 1980) << 9) | (month << 5) | day),
|
||||||
|
.time = @intCast((hour << 11) | (minute << 5) | (second / 2)),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "FAT date/time <-> Unix epoch round trip" {
|
||||||
|
// Even-second UTC times (FAT stores seconds/2, so even seconds round-trip exactly).
|
||||||
|
for ([_]u64{ 1_577_836_800, 1_700_000_000, 1_262_304_000, 1_783_971_244 }) |epoch| {
|
||||||
|
const fat = epochToFatDateTime(epoch);
|
||||||
|
try std.testing.expectEqual(epoch, fatToEpoch(fat.date, fat.time));
|
||||||
|
}
|
||||||
|
// Absolute check: 1577836800 is 2020-01-01 00:00:00 UTC.
|
||||||
|
const y2020 = epochToFatDateTime(1_577_836_800);
|
||||||
|
try std.testing.expectEqual(@as(u16, 2020), 1980 + (y2020.date >> 9));
|
||||||
|
try std.testing.expectEqual(@as(u16, 1), (y2020.date >> 5) & 0x0F); // month
|
||||||
|
try std.testing.expectEqual(@as(u16, 1), y2020.date & 0x1F); // day
|
||||||
|
// 0 is "unset" both ways.
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), fatToEpoch(0, 0));
|
||||||
|
try std.testing.expectEqual(@as(u16, 0), epochToFatDateTime(0).date);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "on-disk struct sizes match the specification" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 36), @sizeOf(BiosParameterBlock));
|
||||||
|
try std.testing.expectEqual(@as(usize, 26), @sizeOf(ExtendedBootRecord16));
|
||||||
|
try std.testing.expectEqual(@as(usize, 54), @sizeOf(ExtendedBootRecord32));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(DirectoryEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(LongNameEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 512), @sizeOf(FileSystemInformation));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "directory entry cluster split/join" {
|
||||||
|
var entry = std.mem.zeroes(DirectoryEntry);
|
||||||
|
entry.setFirstCluster(0x01234567);
|
||||||
|
try std.testing.expectEqual(@as(u16, 0x4567), entry.first_cluster_low);
|
||||||
|
try std.testing.expectEqual(@as(u16, 0x0123), entry.first_cluster_high);
|
||||||
|
try std.testing.expectEqual(@as(u32, 0x01234567), entry.firstCluster());
|
||||||
|
}
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
||||||
//! acpi service (docs/m19-m20-plan.md decision 7). **Placeholder: not
|
//! acpi service (docs/discovery.md — firmware neutrality). **Placeholder: not
|
||||||
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
||||||
//! values from day one; the implementation lands with the Raspberry Pi
|
//! values from day one; the implementation lands with the Raspberry Pi
|
||||||
//! bring-up (docs/arm.md).
|
//! bring-up (docs/arm.md).
|
||||||
@@ -15,7 +15,7 @@
|
|||||||
//! resident under the manager's supervision (hello, restart, the usual
|
//! resident under the manager's supervision (hello, restart, the usual
|
||||||
//! contract).
|
//! contract).
|
||||||
//!
|
//!
|
||||||
//! Known prerequisite recorded in the plan: `DeviceDescriptor`'s 8-byte `hid`
|
//! Known prerequisite recorded in docs/discovery.md: `DeviceDescriptor`'s 8-byte `hid`
|
||||||
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
||||||
//! widens before this file grows a body.
|
//! widens before this file grows a body.
|
||||||
|
|
||||||
|
|||||||
+100
-18
@@ -21,16 +21,31 @@ const std = @import("std");
|
|||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
const power = runtime.power_protocol;
|
const power = runtime.power_protocol;
|
||||||
|
|
||||||
|
/// Where the kernel boot log is persisted on the USB FAT volume — an 8.3 name at
|
||||||
|
/// the mount root (see system/services/log-flush). init writes it at shutdown;
|
||||||
|
/// the log-flush one-shot writes it once at boot.
|
||||||
|
const log_path = "/mnt/usb/DANOS.LOG";
|
||||||
|
|
||||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||||
/// manifest under /system/services instead of a hardcoded list.)
|
/// manifest under /system/services instead of a hardcoded list.)
|
||||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager" };
|
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display", "display-demo" };
|
||||||
|
|
||||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
/// The live process id of each boot service (0 = not running), indexed by its position
|
||||||
var child_count: usize = 0;
|
/// in `boot_services`, plus how many times init has restarted it. init supervises these:
|
||||||
|
/// it spawns them against `supervision_endpoint` and, on a child's death, restarts it (up
|
||||||
|
/// to `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md), the
|
||||||
|
/// service-level counterpart to the device manager's driver restarts.
|
||||||
|
var child_ids: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||||
|
var restart_counts: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||||
|
var shutting_down = false;
|
||||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||||
|
|
||||||
|
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||||
|
/// that faults immediately on every spawn doesn't respawn forever.
|
||||||
|
const maximum_restarts = 3;
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||||
// mmaps pages from the kernel and carves them with the free list), write into
|
// mmaps pages from the kernel and carves them with the free list), write into
|
||||||
@@ -40,7 +55,7 @@ pub fn main() void {
|
|||||||
// the extern malloc/free symbols; Zig code uses this allocator.)
|
// the extern malloc/free symbols; Zig code uses this allocator.)
|
||||||
const gpa = runtime.allocator();
|
const gpa = runtime.allocator();
|
||||||
if (gpa.alloc(u8, 64)) |buffer| {
|
if (gpa.alloc(u8, 64)) |buffer| {
|
||||||
const message = "init: heap ok\n";
|
const message = "/system/services/init: heap ok\n";
|
||||||
@memcpy(buffer[0..message.len], message);
|
@memcpy(buffer[0..message.len], message);
|
||||||
_ = runtime.system.write(buffer[0..message.len]);
|
_ = runtime.system.write(buffer[0..message.len]);
|
||||||
gpa.free(buffer);
|
gpa.free(buffer);
|
||||||
@@ -50,7 +65,7 @@ pub fn main() void {
|
|||||||
// notifications (they are spawned supervised against it), init's own
|
// notifications (they are spawned supervised against it), init's own
|
||||||
// signals, and power events it subscribes to. All arrive in the loop below.
|
// signals, and power events it subscribes to. All arrive in the loop below.
|
||||||
supervision_endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
supervision_endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
||||||
_ = runtime.system.write("init: no endpoint\n");
|
_ = runtime.system.write("/system/services/init: no endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
_ = runtime.process.bindSignals(supervision_endpoint);
|
_ = runtime.process.bindSignals(supervision_endpoint);
|
||||||
@@ -58,13 +73,18 @@ pub fn main() void {
|
|||||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||||
// Best-effort and silent: each service announces its own readiness, and in
|
// Best-effort and silent: each service announces its own readiness, and in
|
||||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||||
for (boot_services) |service| {
|
for (boot_services, 0..) |service, i| {
|
||||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
|
||||||
children[child_count] = id;
|
|
||||||
child_count += 1;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
||||||
|
// volume (/mnt/usb/DANOS.LOG) so it can be read on another machine — the only
|
||||||
|
// way to see it on a headless/real board with no host capturing serial. Fire
|
||||||
|
// and forget: it polls for the mount itself, and is deliberately NOT one of
|
||||||
|
// init's supervised children (a transient one-shot must not be stopped-and-
|
||||||
|
// waited-for during shutdown).
|
||||||
|
_ = runtime.system.spawn("log-flush");
|
||||||
|
|
||||||
// Subscribe to power events (retry: the power service registers well after
|
// Subscribe to power events (retry: the power service registers well after
|
||||||
// init starts). Best-effort — without it, a `terminate` signal still
|
// init starts). Best-effort — without it, a `terminate` signal still
|
||||||
// triggers the same shutdown path.
|
// triggers the same shutdown path.
|
||||||
@@ -83,7 +103,7 @@ pub fn main() void {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (got.isTimer()) {
|
if (got.isTimer()) {
|
||||||
_ = runtime.system.write("init: heartbeat\n");
|
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -92,11 +112,47 @@ pub fn main() void {
|
|||||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// Child-exit notifications and anything else: keep waiting.
|
if (got.isChildExit()) {
|
||||||
|
restartChild(got.childProcessId());
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// Anything else: keep waiting.
|
||||||
if (got.isNotification()) continue;
|
if (got.isNotification()) continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A supervised boot service died. Find which one and restart it — unless it exited
|
||||||
|
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
|
||||||
|
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
||||||
|
/// iron rule 1); init only decides whether to bring it back.
|
||||||
|
fn restartChild(id: u32) void {
|
||||||
|
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||||
|
for (boot_services, 0..) |service, i| {
|
||||||
|
if (child_ids[i] != id) continue;
|
||||||
|
child_ids[i] = 0;
|
||||||
|
// An unknown reason (the record aged out) is treated as a crash worth restarting.
|
||||||
|
const reason = runtime.process.exitReason(id) orelse .fault;
|
||||||
|
if (reason == .exited) {
|
||||||
|
logLine("/system/services/init: {s} exited cleanly; not restarting\n", .{service});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
restart_counts[i] += 1;
|
||||||
|
if (restart_counts[i] > maximum_restarts) {
|
||||||
|
logLine("/system/services/init: {s} keeps crashing; giving up after {d} restarts\n", .{ service, maximum_restarts });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
logLine("/system/services/init: {s} died ({s}); restarting ({d}/{d})\n", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||||
|
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||||
|
}
|
||||||
|
|
||||||
|
fn logLine(comptime fmt: []const u8, args: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, args) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||||
/// call's capability) so events arrive as buffered messages here.
|
/// call's capability) so events arrive as buffered messages here.
|
||||||
fn subscribePower() void {
|
fn subscribePower() void {
|
||||||
@@ -115,15 +171,41 @@ fn subscribePower() void {
|
|||||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The stop sequence: terminate each child in reverse spawn order (vfs last —
|
/// Copy the whole kernel log to /mnt/usb/DANOS.LOG (the same file log-flush
|
||||||
/// other services may flush through it), waiting up to a deadline for each to
|
/// writes at boot), so a poweroff captures the fullest log. Best-effort: if the
|
||||||
/// exit before killing it, then ask the power service to enter S5.
|
/// USB volume is not mounted, the open fails and it does nothing. Must run while
|
||||||
|
/// the storage services are still alive (see shutDown).
|
||||||
|
fn flushKernelLog() void {
|
||||||
|
// Truncate on open so this fuller flush replaces the boot-time one cleanly.
|
||||||
|
var file = runtime.fs.open(log_path, .{ .create = true, .truncate = true }) orelse return; // no USB volume
|
||||||
|
defer file.close();
|
||||||
|
var chunk: [4096]u8 = undefined;
|
||||||
|
var offset: usize = 0;
|
||||||
|
while (true) {
|
||||||
|
const got = runtime.system.klogRead(offset, &chunk);
|
||||||
|
if (got == 0) break; // reached the end of the accumulated log
|
||||||
|
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||||
|
offset += got;
|
||||||
|
}
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, "/system/services/init: flushed log to {s} ({d} bytes)\n", .{ log_path, offset }) catch "");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The stop sequence: persist the log while storage is still up, then terminate
|
||||||
|
/// each child in reverse spawn order (vfs last — other services may flush through
|
||||||
|
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
||||||
|
/// power service to enter S5.
|
||||||
fn shutDown() void {
|
fn shutDown() void {
|
||||||
_ = runtime.system.write("init: shutting down\n");
|
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
|
||||||
var i = child_count;
|
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||||
|
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
||||||
|
// reverse-order stop loop below kills the fat server first, so /mnt/usb must be
|
||||||
|
// written while it is still mounted.
|
||||||
|
flushKernelLog();
|
||||||
|
var i = boot_services.len;
|
||||||
while (i > 0) {
|
while (i > 0) {
|
||||||
i -= 1;
|
i -= 1;
|
||||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
if (child_ids[i] != 0) runtime.process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||||
}
|
}
|
||||||
if (runtime.ipc.lookup(.power)) |h| {
|
if (runtime.ipc.lookup(.power)) |h| {
|
||||||
const request = power.Shutdown{};
|
const request = power.Shutdown{};
|
||||||
|
|||||||
@@ -115,14 +115,14 @@ fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
|
|||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||||
_ = system.write("input: no endpoint\n");
|
_ = system.write("/system/services/input: no endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
if (!ipc.register(.input, endpoint)) {
|
if (!ipc.register(.input, endpoint)) {
|
||||||
_ = system.write("input: register failed\n");
|
_ = system.write("/system/services/input: register failed\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
_ = system.write("input: ready\n");
|
_ = system.write("/system/services/input: ready\n");
|
||||||
|
|
||||||
var reply_buffer: [protocol.reply_size]u8 = undefined;
|
var reply_buffer: [protocol.reply_size]u8 = undefined;
|
||||||
var reply_len: usize = 0;
|
var reply_len: usize = 0;
|
||||||
|
|||||||
@@ -0,0 +1,67 @@
|
|||||||
|
//! system/services/log-flush — a one-shot that copies the kernel's in-memory
|
||||||
|
//! diagnostic log to a file on the mounted USB FAT volume, so the boot log
|
||||||
|
//! survives to be read on another machine. On a headless or real board there is
|
||||||
|
//! no host capturing serial, so without this the log is lost at power-off; this
|
||||||
|
//! is the on-disk equivalent of QEMU's `-serial file:`.
|
||||||
|
//!
|
||||||
|
//! It reads the whole kernel log back through `klog_read` (the RAM sink in
|
||||||
|
//! system/kernel/log.zig) and writes it to /mnt/usb/DANOS.LOG. The name is 8.3
|
||||||
|
//! (FAT short-name rule: base <= 8, extension <= 3) and lives at the mount root
|
||||||
|
//! (there is no mkdir on the FAT path yet). init spawns this once the boot
|
||||||
|
//! services are up; init itself repeats the flush at shutdown for a fuller log.
|
||||||
|
//!
|
||||||
|
//! If no USB volume is mounted — no stick, or the initial-ramdisk sweep that
|
||||||
|
//! spawns every bundled binary bare with no VFS — it waits briefly, then exits
|
||||||
|
//! silently, deranging no other test's output.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const fs = runtime.fs;
|
||||||
|
|
||||||
|
const log_path = "/mnt/usb/DANOS.LOG";
|
||||||
|
|
||||||
|
/// Copy the whole kernel log to the open file, looping klog_read -> write until
|
||||||
|
/// the log is exhausted. Returns the number of bytes written.
|
||||||
|
fn drainKernelLog(file: *fs.File) usize {
|
||||||
|
var chunk: [4096]u8 = undefined;
|
||||||
|
var offset: usize = 0;
|
||||||
|
while (true) {
|
||||||
|
const got = runtime.system.klogRead(offset, &chunk);
|
||||||
|
if (got == 0) break; // reached the end of the accumulated log
|
||||||
|
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||||
|
offset += got;
|
||||||
|
}
|
||||||
|
return offset;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
// Wait for the fat server to mount /mnt/usb (it must bring up the whole USB
|
||||||
|
// storage chain first, so it races us at boot). Bounded: if the mount never
|
||||||
|
// appears — no volume, or the no-VFS ramdisk sweep — give up silently.
|
||||||
|
var ready = false;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (tries < 1400) : (tries += 1) {
|
||||||
|
if (fs.openDirectory("/mnt/usb")) |directory| {
|
||||||
|
var dir = directory;
|
||||||
|
dir.close();
|
||||||
|
ready = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
runtime.system.sleep(50);
|
||||||
|
}
|
||||||
|
if (!ready) return; // /mnt/usb never became available — nothing to persist to
|
||||||
|
|
||||||
|
// Truncate on open: each flush replaces the file, so a shorter log on a later
|
||||||
|
// boot of the same stick leaves no stale tail from a previous, longer one.
|
||||||
|
var file = fs.open(log_path, .{ .create = true, .truncate = true }) orelse return;
|
||||||
|
const written = drainKernelLog(&file);
|
||||||
|
file.close();
|
||||||
|
|
||||||
|
var line: [96]u8 = undefined;
|
||||||
|
_ = runtime.system.write(std.fmt.bufPrint(&line, "log-flush: wrote {d} bytes to {s}\n", .{ written, log_path }) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -1,8 +1,8 @@
|
|||||||
//! The power protocol (docs/m21-plan.md): system power's domain-named surface,
|
//! The power protocol (docs/power.md): system power's domain-named surface,
|
||||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
||||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
||||||
//! learn which firmware they are on (m19-m20-plan.md decision 7). The
|
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
|
||||||
//! vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||||
|
|
||||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||||
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
//! system/services/shm-client — the creating half of the shm test (docs/display-v2.md V2).
|
||||||
|
//! It `shm_create`s a shared region, writes a known pattern into it, and hands the region's
|
||||||
|
//! capability to `shm-server` as an `ipc_call` send_cap. The server maps that capability and
|
||||||
|
//! confirms the pattern is visible — proving cross-process shared memory over the extended
|
||||||
|
//! capability-passing path.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const system = runtime.system;
|
||||||
|
const shm = runtime.shm;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
|
||||||
|
const pattern_len = 4096;
|
||||||
|
|
||||||
|
/// The pattern the server checks — must match shm-server.zig.
|
||||||
|
fn expected(i: usize) u8 {
|
||||||
|
return @truncate(i *% 7 +% 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lookupServer() ?ipc.Handle {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.shm_test)) |h| return h;
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
const region = shm.create(pattern_len) orelse {
|
||||||
|
_ = system.write("shm: create failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < pattern_len) : (i += 1) region.ptr[i] = expected(i);
|
||||||
|
|
||||||
|
const server = lookupServer() orelse {
|
||||||
|
_ = system.write("shm: no server\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
// A non-empty message (so it reaches on_message, not the ping path), carrying the shm
|
||||||
|
// region's capability. The reply is empty; we just need the round trip.
|
||||||
|
var reply: [64]u8 = undefined;
|
||||||
|
_ = ipc.callCap(server, "shm", &reply, region.handle) catch {
|
||||||
|
_ = system.write("shm: call failed\n");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
//! system/services/shm-server — the receiving half of the shm test (docs/display-v2.md V2).
|
||||||
|
//! It registers under `ServiceId.shm_test`; when `shm-client` calls it carrying a
|
||||||
|
//! shared-memory capability, it `shm_map`s that capability and checks the client's pattern
|
||||||
|
//! is visible through the mapping — proving the two processes share the same physical pages
|
||||||
|
//! (not a copy). On success it prints `shm: shared 4096 bytes ok`, the test's marker.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
const system = runtime.system;
|
||||||
|
const shm = runtime.shm;
|
||||||
|
const ipc = runtime.ipc;
|
||||||
|
|
||||||
|
const pattern_len = 4096;
|
||||||
|
|
||||||
|
/// The pattern the client writes — must match shm-client.zig.
|
||||||
|
fn expected(i: usize) u8 {
|
||||||
|
return @truncate(i *% 7 +% 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = message;
|
||||||
|
_ = reply;
|
||||||
|
_ = sender;
|
||||||
|
const cap = capability orelse {
|
||||||
|
_ = system.write("shm: shared FAILED (no capability)\n");
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
const ptr = shm.map(cap) orelse {
|
||||||
|
_ = system.write("shm: shared FAILED (map)\n");
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < pattern_len) : (i += 1) {
|
||||||
|
if (ptr[i] != expected(i)) {
|
||||||
|
_ = system.write("shm: shared FAILED (mismatch)\n");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = system.write("shm: shared 4096 bytes ok\n");
|
||||||
|
return 0; // empty reply — the client only needs the round trip to unblock
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
runtime.service.run(64, .{ .service = .shm_test, .on_message = onMessage });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start;
|
||||||
|
}
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
//! Pure path utilities for the VFS mount router — no IPC, no state, so they are
|
||||||
|
//! host-testable in isolation. The router uses these to decide whether an opened
|
||||||
|
//! path lies under a mount point and, if so, what it looks like relative to that
|
||||||
|
//! mount.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by a
|
||||||
|
/// path separator — return the path relative to the mount ("/" for an exact
|
||||||
|
/// match, otherwise the tail beginning with '/'). Returns null when `path` is not
|
||||||
|
/// under the mount, so a prefix like "/mnt/usb" never captures "/mnt/usbextra".
|
||||||
|
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||||
|
if (path.len < mount_prefix.len) return null;
|
||||||
|
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||||
|
if (path.len == mount_prefix.len) return "/";
|
||||||
|
if (path[mount_prefix.len] != '/') return null;
|
||||||
|
return path[mount_prefix.len..];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `path` is absolute (rooted at '/'). Bare names — what the flat ramfs
|
||||||
|
/// uses — are relative and never route through a mount.
|
||||||
|
pub fn isAbsolute(path: []const u8) bool {
|
||||||
|
return path.len > 0 and path[0] == '/';
|
||||||
|
}
|
||||||
|
|
||||||
|
test "underMount matches only at path boundaries" {
|
||||||
|
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
||||||
|
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
||||||
|
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null); // not a boundary
|
||||||
|
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null); // shorter than the prefix
|
||||||
|
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
||||||
|
try std.testing.expect(underMount("greeting", "/mnt/usb") == null); // a bare name
|
||||||
|
}
|
||||||
|
|
||||||
|
test "isAbsolute distinguishes paths from bare names" {
|
||||||
|
try std.testing.expect(isAbsolute("/mnt/usb"));
|
||||||
|
try std.testing.expect(!isAbsolute("greeting"));
|
||||||
|
try std.testing.expect(!isAbsolute(""));
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user