Compare commits
54
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f | ||
|
|
67702fa250 | ||
|
|
8a38540312 | ||
|
|
54635eecf5 | ||
|
|
184d90c2c6 | ||
|
|
a32eed877d | ||
|
|
347a041d85 | ||
|
|
53e42837e0 | ||
|
|
f52c591f5e | ||
|
|
77d2e22ed1 | ||
|
|
a64a01a6a9 | ||
|
|
35e8921de8 | ||
|
|
3fb9d5936a | ||
|
|
452080e997 | ||
|
|
5b63a841ba | ||
|
|
4b9507bd59 | ||
|
|
7dec1b0767 | ||
|
|
45b8fd8614 | ||
|
|
dd22bfbc48 | ||
|
|
6e60daed6a | ||
|
|
446f655c69 | ||
|
|
a785efa4a3 | ||
|
|
767a2a9a7c | ||
|
|
dd044fb115 | ||
|
|
3ec14509a0 | ||
|
|
1f2c60b3ec | ||
|
|
d71a5f25d3 | ||
|
|
2a0f17ae86 | ||
|
|
9ef61a0844 | ||
|
|
688b9101e8 | ||
|
|
1d7ba814dc | ||
|
|
8aba86b4ce | ||
|
|
a0c83f4b3f | ||
|
|
1ea48ed5d6 | ||
|
|
dfc7d6a609 | ||
|
|
77a3ccd33d | ||
|
|
07da27dc39 | ||
|
|
d89657d0a4 | ||
|
|
849b4b62d4 | ||
|
|
8589bf713b | ||
|
|
738f6aa697 | ||
|
|
01e56e3f36 | ||
|
|
d5d15cefcb | ||
|
|
fd96a35eb9 | ||
|
|
e3fe3f3f45 | ||
|
|
60da667b42 | ||
|
|
36145e623b | ||
|
|
565415327d | ||
|
|
bf6bdb389d |
@@ -82,6 +82,10 @@ straight into CI.
|
||||
Design notes explaining *why* behind the code live in
|
||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||
|
||||
For the hardware needed to run DanOS — minimum specs plus a plain-language guide
|
||||
matching Intel/AMD CPU generations by name — see
|
||||
[`docs/system-requirements.md`](docs/system-requirements.md).
|
||||
|
||||
## Logo
|
||||
|
||||
San Serif Text "Dan OS" with a black karate belt around it.
|
||||
|
||||
+15
-6
@@ -2,6 +2,7 @@ const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
const EdidActive = uefi.protocol.edid.Active;
|
||||
@@ -29,7 +30,7 @@ pub fn main() uefi.Status {
|
||||
// report the reason (boot services are still up) and park the machine so the
|
||||
// message stays on screen.
|
||||
boot() catch |err| {
|
||||
log("\r\ndanos: boot failed: ");
|
||||
log("\r\nEFI: boot failed: ");
|
||||
logBytes(@errorName(err));
|
||||
log("\r\n");
|
||||
while (true) asm volatile ("hlt");
|
||||
@@ -65,14 +66,14 @@ fn boot() !noreturn {
|
||||
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("danos: no /system/services/init (");
|
||||
log("EFI: no /system/services/init (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("danos: no initial_ramdisk (");
|
||||
log("EFI: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
@@ -84,7 +85,7 @@ fn boot() !noreturn {
|
||||
// the map and exiting would invalidate the map key.
|
||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||
|
||||
log("danos: kernel loaded, exiting boot services\r\n");
|
||||
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||
boot_information.memory_map = try exitBootServices(bs);
|
||||
|
||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||
@@ -395,7 +396,7 @@ fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
log("danos: /system/services/init loaded\r\n");
|
||||
progress("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
@@ -403,7 +404,7 @@ fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInfo
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
log("danos: initial_ramdisk loaded\r\n");
|
||||
progress("EFI: initial_ramdisk loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
@@ -561,6 +562,14 @@ fn log(comptime message: []const u8) void {
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||
}
|
||||
|
||||
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||
/// `log` directly and always show, so a failed boot still explains itself.
|
||||
fn progress(comptime message: []const u8) void {
|
||||
if (!build_options.serial) return;
|
||||
log(message);
|
||||
}
|
||||
|
||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||
fn logBytes(bytes: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
|
||||
@@ -58,7 +58,6 @@ fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
posix_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
@@ -78,9 +77,6 @@ fn addUserBinary(
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// POSIX/C compatibility layer, available to any program that wants it
|
||||
// (danos-native code uses `runtime` directly). See library/posix/.
|
||||
.{ .name = "posix", .module = posix_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
@@ -100,6 +96,105 @@ fn addUserBinary(
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||
const KernelModules = struct {
|
||||
boot_handoff: *std.Build.Module,
|
||||
abi: *std.Build.Module,
|
||||
device_abi: *std.Build.Module,
|
||||
architecture: *std.Build.Module,
|
||||
platform: *std.Build.Module,
|
||||
parameters: *std.Build.Module,
|
||||
initial_ramdisk: *std.Build.Module,
|
||||
};
|
||||
|
||||
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||
/// build option baked into `build_options`.
|
||||
fn addKernel(
|
||||
b: *std.Build,
|
||||
kernel_target: std.Build.ResolvedTarget,
|
||||
optimize: std.builtin.OptimizeMode,
|
||||
modules: KernelModules,
|
||||
test_case: ?[]const u8,
|
||||
serial: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
build_options.addOption(bool, "serial", serial);
|
||||
const build_options_module = build_options.createModule();
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||
.{ .name = "abi", .module = modules.abi },
|
||||
.{ .name = "device-abi", .module = modules.device_abi },
|
||||
.{ .name = "architecture", .module = modules.architecture },
|
||||
.{ .name = "platform", .module = modules.platform },
|
||||
.{ .name = "parameters", .module = modules.parameters },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||
/// bundle its own kernel while sharing the (serial-independent) loader, init, and
|
||||
/// ramdisk. Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
init_bin: std.Build.LazyPath,
|
||||
initial_ramdisk_img: std.Build.LazyPath,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_bin);
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
return fat_image;
|
||||
}
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
ensureZigVersion();
|
||||
|
||||
@@ -132,8 +227,8 @@ pub fn build(b: *std.Build) void {
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||
// same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig,
|
||||
// no kernel imports — one source, two builds.
|
||||
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||
// Pure Zig, no kernel imports — one source, two builds.
|
||||
const aml_module = b.addModule("aml", .{
|
||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||
});
|
||||
@@ -142,6 +237,28 @@ pub fn build(b: *std.Build) void {
|
||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||
});
|
||||
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
// class requests, descriptors) and the USB class-code taxonomy — the flat
|
||||
// reference the xHCI bus driver, the USB class drivers, and the device
|
||||
// manager's identity matcher all share. Pure data, like pci-class/acpi-ids.
|
||||
const usb_abi_module = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("system/devices/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids_module = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("system/devices/usb-ids.zig"),
|
||||
});
|
||||
// The USB transfer protocol: what a USB class driver says to the xHCI bus
|
||||
// driver to drive its device (open / control / interrupt / bulk). A protocol
|
||||
// module like vfs-protocol, shared by the bus driver and every class driver.
|
||||
const usb_transfer_protocol_module = b.addModule("usb-transfer-protocol", .{
|
||||
.root_source_file = b.path("system/drivers/usb-xhci-bus/usb-transfer-protocol.zig"),
|
||||
});
|
||||
// The block-device protocol: read/write of fixed-size blocks, spoken between a
|
||||
// filesystem and a block driver (usb-storage). A protocol module like the rest.
|
||||
const block_protocol_module = b.addModule("block-protocol", .{
|
||||
.root_source_file = b.path("system/services/block/protocol.zig"),
|
||||
});
|
||||
|
||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||
@@ -225,6 +342,17 @@ pub fn build(b: *std.Build) void {
|
||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The USB transfer protocol, so runtime.usb (the class-driver client) can speak
|
||||
// it, the way runtime.input speaks the input protocol.
|
||||
runtime_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||
@@ -247,18 +375,6 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
|
||||
// The POSIX / C compatibility layer, a separate library layered strictly over the
|
||||
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
|
||||
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
|
||||
// and library/posix/posix.zig.
|
||||
const posix_module = b.addModule("posix", .{
|
||||
.root_source_file = b.path("library/posix/posix.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
@@ -268,9 +384,12 @@ pub fn build(b: *std.Build) void {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
const build_options_module = build_options.createModule();
|
||||
// The serial-console log sink. Off by default: a real machine often has no
|
||||
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||
|
||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||
@@ -282,41 +401,17 @@ pub fn build(b: *std.Build) void {
|
||||
.abi = .none,
|
||||
});
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "architecture", .module = architecture_module },
|
||||
.{ .name = "platform", .module = platform_module },
|
||||
.{ .name = "parameters", .module = parameters_module },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
const kernel_modules = KernelModules{
|
||||
.boot_handoff = boot_handoff_module,
|
||||
.abi = abi_module,
|
||||
.device_abi = device_abi_module,
|
||||
.architecture = architecture_module,
|
||||
.platform = platform_module,
|
||||
.parameters = parameters_module,
|
||||
.initial_ramdisk = initial_ramdisk_module,
|
||||
};
|
||||
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
@@ -330,7 +425,7 @@ pub fn build(b: *std.Build) void {
|
||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
@@ -338,21 +433,43 @@ pub fn build(b: *std.Build) void {
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||
// the input service. They build chapter-9 class requests from usb-abi.
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
usb_hid_keyboard_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
usb_hid_mouse_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
usb_storage_exe.root_module.addImport("block-protocol", block_protocol_module);
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||
// flattened device tree — the aarch64 target flips the default when it
|
||||
@@ -364,16 +481,22 @@ pub fn build(b: *std.Build) void {
|
||||
.acpi => "system/services/acpi/acpi.zig",
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
device_manager_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
@@ -385,10 +508,6 @@ pub fn build(b: *std.Build) void {
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("hpet");
|
||||
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
||||
mk_run.addArg("bus");
|
||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
@@ -397,6 +516,16 @@ pub fn build(b: *std.Build) void {
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-keyboard");
|
||||
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-mouse");
|
||||
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-storage");
|
||||
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||
mk_run.addArg("fat");
|
||||
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||
mk_run.addArg("fat-test");
|
||||
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
@@ -417,6 +546,8 @@ pub fn build(b: *std.Build) void {
|
||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||
mk_run.addArg("process-test");
|
||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||
mk_run.addArg("log-flush");
|
||||
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
@@ -424,12 +555,15 @@ pub fn build(b: *std.Build) void {
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ hpet_exe, "system/drivers" },
|
||||
.{ bus_exe, "system/drivers" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||
.{ usb_storage_exe, "system/drivers" },
|
||||
.{ fat_exe, "system/services" },
|
||||
.{ log_flush_exe, "system/services" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
@@ -442,6 +576,13 @@ pub fn build(b: *std.Build) void {
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||
// which firmware may mirror to a serial console) are silenced by default — a
|
||||
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||
const loader_options = b.addOptions();
|
||||
loader_options.addOption(bool, "serial", serial);
|
||||
const loader_options_module = loader_options.createModule();
|
||||
const efiexe = b.addExecutable(.{
|
||||
.name = "BOOTX64",
|
||||
.root_module = b.createModule(.{
|
||||
@@ -454,6 +595,7 @@ pub fn build(b: *std.Build) void {
|
||||
.imports = &.{
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "build_options", .module = loader_options_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -463,6 +605,32 @@ pub fn build(b: *std.Build) void {
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
check_fat.addArg("--verify");
|
||||
check_fat.addFileArg(fat_image);
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
@@ -523,10 +691,14 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Present the FHS zig-out to the guest as a FAT drive — it is the boot volume.
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-drive",
|
||||
b.fmt("format=raw,file=fat:rw:{s}", .{b.install_path}),
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
@@ -545,8 +717,10 @@ pub fn build(b: *std.Build) void {
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||
run_efi.step.dependOn(b.getInstallStep());
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
@@ -572,11 +746,19 @@ pub fn build(b: *std.Build) void {
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
}) |root| {
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
@@ -603,6 +785,21 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||
// loop above.
|
||||
const time_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/time.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||
|
||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||
|
||||
+33
-14
@@ -82,9 +82,19 @@ Start with the north star:
|
||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||
fault isolation + live restart — the reincarnation-server + capability model that
|
||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
- **[system-requirements.md](system-requirements.md) — system requirements.** The
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
@@ -96,7 +106,16 @@ Cutting across all of these:
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs.
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
@@ -153,7 +172,7 @@ addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
@@ -178,21 +197,22 @@ system/ → /system danos's own internals (the self-representation)
|
||||
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||
vfs.zig, vfs-test.zig, protocol.zig)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime — the stable application ABI
|
||||
posix/ POSIX/C compatibility, layered over runtime
|
||||
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the POSIX
|
||||
layer imports by name. `usb`/`block` drivers will expose their protocols the same way.
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
|
||||
file API (`runtime.fs`) imports by name. `usb`/`block` drivers expose their protocols the
|
||||
same way.
|
||||
|
||||
`library/posix/` is special: it is the **one place** POSIX/C spellings are allowed
|
||||
verbatim (`stat`, `O_CREAT`, `fopen`, `errno`). Everywhere else follows the danos
|
||||
naming rule with no exception — see [coding-standards.md](coding-standards.md). The
|
||||
POSIX layer calls the runtime, never the kernel's system calls directly, so it never
|
||||
appears in the private-ABI path.
|
||||
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||
exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
|
||||
## Source map
|
||||
|
||||
@@ -216,9 +236,8 @@ appears in the private-ABI path.
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
+55
-1
@@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a
|
||||
part that answers "where are the tables?" — everything past the RSDP is just following
|
||||
more pointers the tables themselves provide.
|
||||
|
||||
## ACPI events: the SCI, the power button, and GPEs (M21)
|
||||
|
||||
The tables above are static description; ACPI is also a *live* channel. Hardware
|
||||
raises the **SCI** (System Control Interrupt) — one shared, level-triggered line
|
||||
whose vector the FADT names — and the OS reads status registers to learn what
|
||||
happened: a fixed event like the power button, or a **General-Purpose Event**
|
||||
(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML
|
||||
to ring 3, the event side lives there too, in the same **acpi service** — the
|
||||
device discoverer and the event source are one process, because both need the
|
||||
namespace and the port grant.
|
||||
|
||||
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||
header, while the blob resources are header-stripped bytecode that starts with
|
||||
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||
finds the line to `irq_bind`.
|
||||
|
||||
With those in hand the service enables ACPI mode (only if `SCI_EN` is clear —
|
||||
some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||
|
||||
- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status.
|
||||
The handler clears it (write-1-to-clear), logs the press, and publishes a
|
||||
[`power`](power.md) `power_button` event to subscribers.
|
||||
- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the
|
||||
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||
method, drains the **Notify** queue that method produced, maps each notified
|
||||
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||
per evaluation; everything else a handler needs (field access, control flow,
|
||||
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||
|
||||
**How this is tested.** QEMU cannot raise GPEs deterministically on this config,
|
||||
so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||
are interface-complete but validated on real hardware later.
|
||||
|
||||
The service surface these events are *published on* — subscription, the event
|
||||
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||
|
||||
## Related
|
||||
|
||||
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
||||
the ACPI-reclaim memory the RSDP lives in.
|
||||
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
||||
tables into one neutral device model shared with the ARM device-tree path.
|
||||
tables into one neutral device model shared with the ARM device-tree path, and how
|
||||
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||
module and never names ACPI directly.
|
||||
|
||||
+40
-11
@@ -65,15 +65,18 @@ Three, and only three.
|
||||
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||
`fwrite` any more.
|
||||
|
||||
**This exception is scoped to one place: `library/posix/`.** A file under
|
||||
`library/posix/` *is* the foreign ABI, so it keeps the ABI's spellings — that is the
|
||||
whole rule for that directory. **Everywhere else, Zig/danos naming applies with no
|
||||
POSIX exception**, so there is nothing to get wrong: if you're not in
|
||||
`library/posix/`, expand it. A concept POSIX also has gets a danos name outside that
|
||||
layer — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||
flag, not `O_CREAT`; `library/posix/` is what maps `stat`→`status` and
|
||||
`O_CREAT`→`create` at the boundary. (The `syscall` *wrappers* elsewhere are not an
|
||||
exception to this — they wrap the private danos ABI, so they use danos names.)
|
||||
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||
layer is premature until danos actually needs it (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||
`system` interface, so it keeps `open`/`read`/`errno`/`O_CREAT`. **Everywhere else,
|
||||
Zig/danos naming applies with no exception**: a concept POSIX also has gets a danos
|
||||
name — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||
flag, not `O_CREAT`; the boundary is where `stat`→`status` and `O_CREAT`→`create` get
|
||||
mapped. (The `syscall` *wrappers* elsewhere are not an exception — they wrap the
|
||||
private danos ABI, so they use danos names.)
|
||||
|
||||
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||
not ours, and are left alone:
|
||||
@@ -96,7 +99,7 @@ That's all — no Unix-abbreviation exception. The source directories are full w
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
||||
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
## A note on collisions
|
||||
@@ -138,10 +141,36 @@ single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
||||
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
|
||||
## Named values, not magic numbers
|
||||
|
||||
The naming rule has a twin: **a value with meaning gets a name, too.** The same
|
||||
principle drives both — a reader should never have to leave the code to understand it.
|
||||
An abbreviated *name* forces a reader to guess; a bare *number* forces them worse, out
|
||||
to a spec or a header or a comment three files away, to learn what the value even *is*.
|
||||
If `0x0C` is the PCI serial-bus class, the code says `BaseClass.serial_bus`, not `0x0C`;
|
||||
if `0x04` is the ACPI IRQ resource descriptor, it says `SmallResourceType.irq`, not
|
||||
`0x04`. The number is an implementation detail of the name — recorded once, where the
|
||||
name is defined, and never spelled again at a use site.
|
||||
|
||||
**Prefer an `enum`** when the values form a set (device classes, AML opcodes, resource
|
||||
descriptor types, states): the type then also says *which* set a value belongs to, and
|
||||
the compiler rejects a value from the wrong one. A lone `pub const` with a descriptive
|
||||
name suffices for a one-off (`const large_descriptor_bit = 0x80`). Reach for the enum
|
||||
the moment code elsewhere compares against, packs, or produces the value — a packed PCI
|
||||
class triple is written from named parts (`.serial_bus`, `.usb`, `.xhci`), never as
|
||||
`0x0C_03_30` under a comment that decodes the bytes.
|
||||
|
||||
The exceptions are the numbers that carry no hidden meaning: `0` and `1` as plain zero
|
||||
and one, an index step, a field width, a bit shift. `x + 1`, `buffer[0]`, and `<< 8`
|
||||
need no christening — there is nothing to look up. The test is exactly the naming test:
|
||||
*would a reader have to look this up to know what it means?* If yes, name it. This is
|
||||
what `opcodes.zig`'s `*_opcode` constants, `acpi-ids`'s `HardwareId`, and `pci-class`'s
|
||||
class enums already are — reference data defined once and named everywhere it is used.
|
||||
|
||||
## Why acronyms are the line
|
||||
|
||||
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||
|
||||
@@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
@@ -61,8 +61,8 @@ to the driver in the order written, and a read consumes what is there. Terminals
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
||||
that program, minus the client half.
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||
is already that program, minus the file-node client half.
|
||||
|
||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||
|
||||
@@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea
|
||||
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
||||
timer — a later step.
|
||||
|
||||
### Is the TSC trustworthy? Invariant, and synchronized
|
||||
|
||||
A cycle counter is only a valid *clock* if two things hold, and danos checks both,
|
||||
because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET.
|
||||
|
||||
**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with
|
||||
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||
measured frequency without the invariance guarantee is not enough.
|
||||
|
||||
**Synchronized.** Each core has its own TSC. Even invariant ones can start at different
|
||||
values (a second socket, some firmware), so a thread migrating from a core reading
|
||||
`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor
|
||||
drift with frequency. It costs a memory-mapped read instead of a register read, but it
|
||||
keeps time *accurate*, which is the whole point. The switch preserves the current value,
|
||||
so the clock never jumps. The boot log names the outcome:
|
||||
|
||||
```
|
||||
/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD
|
||||
/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG)
|
||||
```
|
||||
|
||||
## Two kinds of vector, one dispatch
|
||||
|
||||
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
||||
|
||||
+21
-5
@@ -51,7 +51,16 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands.
|
||||
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
instead of appending a duplicate. The kernel table has no unregister, so without
|
||||
this a restarted registering bus would re-report its children as fresh nodes on
|
||||
every respawn. Idempotence is what makes restart-and-re-report sound for *every*
|
||||
reporting bus — pci-bus, the acpi service, a future fdt service — not just one,
|
||||
and it is why supervision (below) can prune a dead bus's subtree and trust the
|
||||
restarted instance to rebuild exactly the same ids.
|
||||
|
||||
## The protocol
|
||||
|
||||
@@ -136,10 +145,17 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
the mouse and keyboard QEMU already hangs off it.
|
||||
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19)
|
||||
then the acpi service (M20) moved enumeration to ring 3; the kernel seeds
|
||||
only the host bridge and the acpi-tables node. See
|
||||
[m19-m20-plan.md](m19-m20-plan.md).
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||
PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3
|
||||
— each in a single phase so no device is ever matched from both sources at
|
||||
once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus
|
||||
already reports PCI functions (M20.2).
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
|
||||
@@ -191,3 +191,57 @@ and registers + reports each `_HID` device — the device manager matches driver
|
||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||
entirely in user space; the kernel seeds only the host bridge and the
|
||||
acpi-tables node.
|
||||
|
||||
## Discovery is a swappable process per firmware (M19–M20)
|
||||
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
|
||||
- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi
|
||||
service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML.
|
||||
- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is
|
||||
an **fdt service**: it claims a `devicetree-blob` node and walks the tree —
|
||||
pure data, no bytecode, so it needs neither a port grant nor an interpreter,
|
||||
strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md)
|
||||
bring-up fills it in.)
|
||||
|
||||
The device manager spawns the discoverer under the **neutral ramdisk name
|
||||
`discovery`** and never learns which firmware it is on; the build's
|
||||
`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the
|
||||
aarch64 target flips the default when it lands). The manager owns the device
|
||||
tree as *data* and touches no hardware, ever — firmware bytecode runs only
|
||||
inside the crashable, supervised discoverer, so an AML fault can never take
|
||||
down the supervisor.
|
||||
|
||||
Two consequences of neutrality bind on later work:
|
||||
|
||||
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||
`ServiceId.power`, and subscribers never learn the difference.
|
||||
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||
report a real node.
|
||||
|
||||
Two supporting decisions keep the kernel's remaining slice honest:
|
||||
|
||||
- **The AML interpreter is a shared build module**, compiled into both the
|
||||
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||
device count across the ring-3 move.
|
||||
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||
PCI functions carry BAR resources, and `device_register` containment demands
|
||||
the bridge own windows that cover them. Those apertures are derived
|
||||
kernel-side from the boot memory map's MMIO holes (regions that are neither
|
||||
RAM nor tables) — mechanical, AML-free, and available at boot regardless of
|
||||
what later moved to user space. The acpi service's authority is likewise
|
||||
exactly one node: the `acpi-tables` node, whose broad io_port grant is the
|
||||
documented trust boundary for the one process allowed to run firmware
|
||||
bytecode.
|
||||
|
||||
+10
-9
@@ -57,8 +57,9 @@ is not an address window. Discovery is trusted; user space is not.
|
||||
|
||||
### What a bus driver looks like
|
||||
|
||||
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||
its "devices" are the block's comparators:
|
||||
danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and
|
||||
`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block
|
||||
as the "bus" and its comparators as the "devices", is:
|
||||
|
||||
```zig
|
||||
_ = dev.claim(bus.id); // 1. own the bus
|
||||
@@ -78,8 +79,8 @@ for (0..n) |i| { // 3. publish each child
|
||||
|
||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
||||
asserts the kernel's table upholds it.
|
||||
whose window escapes the bus is refused; the in-kernel `containment` test asserts the
|
||||
kernel's table upholds that ([drivers.md](drivers.md)).
|
||||
|
||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||
@@ -143,7 +144,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||
no DMA driver consumes `dma_alloc` yet.
|
||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||
@@ -302,8 +303,8 @@ rather than an out-struct. The rest of this section is the original design note.
|
||||
|
||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
||||
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||
device has.
|
||||
|
||||
**The fix, in two halves.**
|
||||
@@ -326,7 +327,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i
|
||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||
@@ -353,7 +354,7 @@ gap should be named rather than implied.
|
||||
|
||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||
could write a real one against `bus`'s comparators tomorrow.
|
||||
could write a real one against any device a bus driver publishes tomorrow.
|
||||
|
||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||
|
||||
+40
-29
@@ -22,12 +22,12 @@ say.*
|
||||
|
||||
## How a driver gets started: discover, match, spawn
|
||||
|
||||
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
||||
and policy lives in user space. Boot brings user space up as a three-level supervision
|
||||
hierarchy, each level owning one job:
|
||||
Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that
|
||||
is policy, and policy lives in user space. Boot brings user space up as a three-level
|
||||
supervision hierarchy, each level owning one job:
|
||||
|
||||
```
|
||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus
|
||||
| | |
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
@@ -184,7 +184,11 @@ Two properties worth knowing:
|
||||
|
||||
## A whole driver
|
||||
|
||||
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
||||
A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such
|
||||
example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`,
|
||||
`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a
|
||||
compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the
|
||||
shape is:
|
||||
|
||||
```zig
|
||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||
@@ -209,8 +213,8 @@ while (...) {
|
||||
}
|
||||
```
|
||||
|
||||
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
||||
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
||||
The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a
|
||||
clocksource — the only way to use it is to read it, so it exercises `mmio_map` without
|
||||
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||
@@ -252,9 +256,10 @@ bus driver may only ever subdivide what it already owns.
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
||||
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||
controller drivers fit together.
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||
drivers and host controller drivers fit together.
|
||||
|
||||
## What the kernel does not do for you
|
||||
|
||||
@@ -313,32 +318,38 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
||||
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
||||
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||
No demo driver ships to prove this end to end; the *real* drivers do, so the tests
|
||||
target them and the kernel primitives directly:
|
||||
|
||||
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||
APIC redirection entry back and asserts the line really is routed to a device vector,
|
||||
really is level-triggered, and really was left unmasked by the driver's final
|
||||
`irq_ack`.
|
||||
- **`device-manager`** — boots only the device manager, which discovers the PCI host
|
||||
bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the
|
||||
process table and the device tree — to confirm pci-bus came up and registered the
|
||||
functions it enumerated: the whole discover → match → spawn → driver-up chain.
|
||||
- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ,
|
||||
delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end.
|
||||
- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM
|
||||
window) and walks it: `mmio_map`, end to end.
|
||||
- **`containment`** — the kernel refuses a `device_register` whose child window escapes
|
||||
the parent's grant (else it would be a syscall for mapping arbitrary memory), while an
|
||||
identical re-register stays idempotent. Asserted in-kernel, straight against the broker.
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
is not. That second half is why bindings are keyed on the owning *task* and not on the
|
||||
endpoint pointer — endpoints are shared, so releasing "everything pointing at this
|
||||
endpoint" would silently mask a live driver's device.
|
||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||
space never returns MMIO frames to the RAM pool.
|
||||
|
||||
```
|
||||
$ python3 test/qemu_test.py hpet irqfree iopass
|
||||
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass
|
||||
device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
acpi-ps2 ... PASS
|
||||
pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
containment ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
```
|
||||
|
||||
Two companions cover what `hpet` can't, because it never exits:
|
||||
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
is not. That second half is why bindings are keyed on the owning *task* and not on
|
||||
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
||||
this endpoint" would silently mask a live driver's device.
|
||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||
space never returns MMIO frames to the RAM pool.
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||
|
||||
@@ -1,210 +0,0 @@
|
||||
# M17–M18 execution plan: process lifecycle + device manager
|
||||
|
||||
**Archived — completed 2026-07-13** (every item checked; suite ended 54/54).
|
||||
Kept as the record of how M17–M18 landed; the successor is
|
||||
[m19-m20-plan.md](m19-m20-plan.md).
|
||||
|
||||
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
||||
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
||||
settled in those documents; this file is the build order — one phase at a time,
|
||||
each phase green before the next starts. Delete or archive this file when M18
|
||||
lands.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
||||
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
||||
green phase (no co-author trailers).
|
||||
|
||||
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
||||
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
||||
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
||||
phases are all green it is **auto-merged into `main`**; branches are kept after
|
||||
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
||||
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
||||
run the existing QEMU suite green before any new work starts.
|
||||
|
||||
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
||||
|
||||
## Status
|
||||
|
||||
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
||||
only when its definition of green holds.
|
||||
|
||||
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
||||
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
||||
suite green from the worktree (48/48, 2026-07-12).
|
||||
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
||||
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
||||
test; suite 49/49)
|
||||
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
||||
the notification; `process_exit_reason` supervisor-gated;
|
||||
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
||||
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
||||
bounded ref-counted table, publish on every death;
|
||||
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
||||
the owner's death; `vfs-client-death` test; suite 50/50)
|
||||
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
||||
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
||||
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
||||
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
||||
scenario; suite 51/51)
|
||||
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
||||
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
||||
(device-manager-protocol module; the manager as a harness service:
|
||||
supervised spawns, hello deadline via timer sweep, restart with
|
||||
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
||||
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
||||
re-proving claim release each respawn; `driver-restart` scenario;
|
||||
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
||||
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
||||
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
||||
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
||||
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
||||
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
||||
scenario proves report → prune → respawn → re-report; suite 53/53)
|
||||
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
||||
endpoint rides as the call's capability; events are the same structs the
|
||||
buses send); device-list first client; protocol capped at the kernel's
|
||||
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
||||
sheared concurrent serial lines and was the scenario-flake root cause;
|
||||
`device-list` scenario; suite 54/54)
|
||||
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
||||
|
||||
---
|
||||
|
||||
## M17.1 — the kernel releases a dead process's claims
|
||||
|
||||
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
||||
|
||||
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
||||
every `claimed[]` slot holding this task id.
|
||||
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
||||
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
||||
slot in after them, before the exit notification).
|
||||
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
||||
module) and release those by owner in the same pass.
|
||||
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
||||
|
||||
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
||||
device, is killed, is respawned, and claims the same device again successfully;
|
||||
assert both claims in the serial log. Kernel-side unit coverage in
|
||||
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
||||
release one owner, verify exactly its claims freed).
|
||||
|
||||
## M17.2 — exit reasons
|
||||
|
||||
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
||||
illegal_instruction, arithmetic_fault, killed).
|
||||
- Kernel: record the reason at every death site — clean exit path, each fault
|
||||
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
||||
reused, so a small ring keyed by id is enough).
|
||||
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
||||
the recorded reason or `-ESRCH` once evicted.
|
||||
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
||||
- Docs: remove the no-exit-status bullet from process-management.md.
|
||||
|
||||
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
||||
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
||||
three reasons.
|
||||
|
||||
## M17.3 — published exit events
|
||||
|
||||
- Kernel: bounded subscriber table (endpoints); new system call
|
||||
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
||||
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
||||
path already uses.
|
||||
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
||||
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
||||
by that task id (badges already are task ids). Log the release.
|
||||
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
||||
badge encoding).
|
||||
|
||||
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
||||
killed without closing; assert the VFS logs the handle release and its open-handle
|
||||
count returns to baseline.
|
||||
|
||||
## M17.4 — signals and the service harness
|
||||
|
||||
- Kernel: per-task pending mask + bound endpoint; system calls
|
||||
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
||||
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
||||
signals with no bound endpoint pend silently.
|
||||
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
||||
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
||||
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
||||
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
||||
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
||||
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
||||
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
||||
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
||||
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
||||
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
||||
gets for free later.
|
||||
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
||||
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
||||
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
||||
`ping` automatically. Define the reserved `ping` request encoding here and
|
||||
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
||||
existing protocols).
|
||||
- Convert one existing service (input-source or hpet) to the harness as proof it
|
||||
subtracts code rather than adding it.
|
||||
|
||||
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
||||
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
||||
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
||||
deadline with reason `killed`.
|
||||
|
||||
## M18.1 — device-manager protocol: hello + restart policy
|
||||
|
||||
- New `system/services/device-manager/device-manager-protocol.zig` module
|
||||
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
||||
reserved fields.
|
||||
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
||||
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
||||
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
||||
restart vs not.
|
||||
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
||||
conversion is mechanical; otherwise they keep working unconverted (the manager
|
||||
only enforces hello on drivers spawned with an assignment).
|
||||
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
||||
|
||||
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
||||
argv flag to fault after hello on its first run; assert: fault, exit reason
|
||||
recorded, manager respawns with backoff, second run claims the controller
|
||||
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
||||
faults (a tiny test driver, not xhci).
|
||||
|
||||
## M18.2 — bus tree reports
|
||||
|
||||
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
||||
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
||||
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
||||
report one `child_added` per connected port with speed + port number as
|
||||
identity. **No transfer rings, no descriptors** — reading device/interface
|
||||
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
||||
track, not this plan.
|
||||
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
||||
`child_removed`) when a bus driver dies; assert re-report on restart.
|
||||
|
||||
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
||||
`child_added` events reach the manager and appear in its tree dump; kill the
|
||||
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
||||
|
||||
## M18.3 — the application surface
|
||||
|
||||
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
||||
events, input-service pattern).
|
||||
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
||||
becomes the one answer to "what devices exist" for user space.
|
||||
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
||||
discovery migration, out of this plan.
|
||||
|
||||
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
||||
during a driver restart the subscribing client logs remove + add events.
|
||||
|
||||
---
|
||||
|
||||
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
||||
driver, acpi service, retiring the kernel scan), USB control transfers +
|
||||
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
||||
(needs a console), job control.
|
||||
@@ -1,212 +0,0 @@
|
||||
# M19–M20 execution plan: discovery migration
|
||||
|
||||
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
||||
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
||||
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
||||
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
||||
next; this file is the build order and the checklist.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
||||
one), and the relevant design doc updated. Commit per green phase (no co-author
|
||||
trailers). The full suite is the regression net — the existing
|
||||
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
||||
green *through* the migration, which is the whole point: the system must not be
|
||||
able to tell who enumerated it.
|
||||
|
||||
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
||||
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
||||
branch is green; keep branches; push everything.
|
||||
|
||||
## Settled decisions (2026-07-13 — veto before the loop starts)
|
||||
|
||||
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
||||
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
||||
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
||||
power tests prove it), and MCFG (the host bridge node). What retires is
|
||||
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
||||
namespace walk that builds device nodes (M20.3). The AML module stays a
|
||||
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
||||
service (for everything else) — same source, two builds, no fork.
|
||||
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||
PCI functions carry BAR resources, and containment demands the bridge own
|
||||
windows that cover them. The apertures are derived kernel-side from the
|
||||
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
||||
mechanical, AML-free, and available at boot regardless of what later moved
|
||||
to user space. (The bridge today carries only ECAM + bus range; this is the
|
||||
prerequisite M19.0 exists for.)
|
||||
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
||||
with identical (parent, class, resources) returns the existing id instead
|
||||
of appending. The kernel table has no unregister, so without this a
|
||||
restarted registering bus would duplicate its children on every respawn —
|
||||
idempotence makes restart-and-re-report safe for every future bus, not just
|
||||
PCI.
|
||||
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
||||
field (the kernel-registered id, `no_device` for unregistered leaves like
|
||||
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
||||
identity (the class triple) instead of the manager's boot-time snapshot —
|
||||
the snapshot match remains only for what the kernel still seeds. One flip
|
||||
phase changes both sides at once so no device is ever matched twice.
|
||||
5. **The acpi service's authority is one node.** The kernel publishes an
|
||||
`acpi-tables` device: memory resources covering the table blobs plus a
|
||||
broad `io_port` resource — the documented trust grant to exactly one
|
||||
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
||||
io_read/io_write calls already exist). The service claims it, maps the
|
||||
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
||||
`mmio_map` + `io_read`/`io_write`.
|
||||
6. **Both new processes are protocol drivers** under the manager: hello,
|
||||
supervision, restart with backoff — all inherited from M18.1 for free.
|
||||
Registration idempotence (decision 3) is what makes their restarts sound.
|
||||
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
||||
everything at and above the device-manager protocol — descriptors,
|
||||
containment, reports, matching, supervision — and none of it may become
|
||||
x86-specific. Discovery is one swappable process per firmware: the acpi
|
||||
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
||||
`devicetree-blob` node, reports children from the flattened device tree —
|
||||
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
||||
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
||||
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
||||
can never take down the supervisor. Two consequences recorded now:
|
||||
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
||||
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
||||
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
||||
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
||||
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
||||
both services exist as placeholders (system/services/acpi, system/services/
|
||||
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
||||
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
||||
in M20.3 and never learn which firmware it is on.
|
||||
|
||||
## Status
|
||||
|
||||
- [x] **M19.0** — prerequisites (bridge apertures from the memory map's
|
||||
*gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB,
|
||||
caught by the new every-BAR-contained assert in `discovery`; idempotent
|
||||
`device_register` proven in `bus`; `ChildAdded.device_id`;
|
||||
m17-m18-plan.md archived; suite 54/54).
|
||||
- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM
|
||||
through its grant, brute-force walk with the multifunction rule; the
|
||||
manager matches pci_host_bridge → pci-bus per device with the full
|
||||
protocol contract; `pci-scan` builds its expected marker from the
|
||||
kernel's own count — equivalence on the first run; suite 55/55).
|
||||
- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the
|
||||
kernel's addBars so dedupe returns the kernel's node ids during
|
||||
coexistence; the bridge gained the io_port aperture I/O BARs need;
|
||||
reports carry the registered device_id; pci-scan drills a forced restart
|
||||
and asserts the PCI node count never grows — plus harness hardening: a
|
||||
failing case now preserves its serial as <case>-failed-serial.log, and
|
||||
the heavy scenarios run at 150s; suite 55/55).
|
||||
- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all
|
||||
deleted (bridge node stays); manager matches PCI drivers from reported
|
||||
identity, deduped by registered id. Surfaced and fixed a real SMP race the
|
||||
flip created — ring-3 device_register made the broker table concurrent, so
|
||||
mmio_map's lock-free read intermittently tore hpet's resource length
|
||||
(user fault) and overflowed `r.len-1` into a kernel panic; now the broker
|
||||
read is under the big lock and the arithmetic is guarded, and pci-bus
|
||||
skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart
|
||||
hammered 6×).
|
||||
- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13).
|
||||
- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build
|
||||
module compiled into both kernel and service; the kernel publishes the
|
||||
`acpi-tables` node (AML blobs as memory resources, the broad io_port grant,
|
||||
the SCI); the service claims it, maps the blobs, runs the shared parser in
|
||||
ring 3, and self-verifies its Device count against the kernel's (34 = 34,
|
||||
deterministic via argv, no log-scraping); the manager spawns `discovery`
|
||||
at startup. Parse-only touches no hardware. Suite 56/56.
|
||||
|
||||
- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in
|
||||
ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page
|
||||
backs SystemMemory maps so a stray region can't fault it) and registers +
|
||||
reports each present `_HID` device under `acpi-tables`. Containment: the
|
||||
broker's irq check became range-based (len-1 == the old equality) so the
|
||||
node's broad irq window covers children's legacy lines; io ports fall in
|
||||
the broad io grant. ChildAdded gained `hid`. Matching stays off. The
|
||||
`acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse
|
||||
(1 resource) among the reports. Suite 57/57.
|
||||
- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the
|
||||
device-building helpers are retained-but-dead pending a focused sweep,
|
||||
spawned as a task; static tables + `\_S5` + the acpi-tables node stay).
|
||||
The manager matches ps2-bus from ACPI `_HID` reports; the service
|
||||
registers all devices before reporting any (no keyboard-before-mouse
|
||||
race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches
|
||||
its keyboard; `ioport` retargeted to the acpi-tables I/O window (the
|
||||
kernel-built PS/2 node is gone). Suite 58/58.
|
||||
- [ ] **merge** `feat/acpi-service` → main, push — **loop ends here**.
|
||||
|
||||
---
|
||||
|
||||
## Phase notes
|
||||
|
||||
**M19.0 apertures:** the boot memory map already crosses the handoff
|
||||
([boot-handoff]), but discovery never sees it today — expect a small
|
||||
pass-through (kernel init hands the map to the platform layer) before the
|
||||
holes computation, which belongs where the bridge node is built
|
||||
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
||||
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
||||
the kernel unit test.
|
||||
|
||||
**M19.1 scanning without owning config access twice:** the driver reads config
|
||||
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
||||
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
||||
no bridge recursion (matches the kernel's current single-segment walk).
|
||||
|
||||
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
||||
restore) is deferred — the BARs' current programmed values and types are
|
||||
enough for containment-checked registration at bring-up; sizing lands with the
|
||||
first driver that needs to *move* a BAR. Log what is registered so the
|
||||
scenario can assert it.
|
||||
|
||||
**M19.3 what the manager still seeds from the snapshot:** everything the
|
||||
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
||||
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
||||
|
||||
**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns
|
||||
`discovery` by its neutral ramdisk name at startup, as an ordinary protocol
|
||||
driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI
|
||||
devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none):
|
||||
firmware *string* identity travels beside the numeric `identity` field until
|
||||
the FDT-driven widening replaces both (decision 7).
|
||||
|
||||
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
||||
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
||||
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
||||
tell it moved — that is the assertion of `acpi-parse`.
|
||||
|
||||
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
||||
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
||||
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
||||
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
||||
|
||||
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
||||
its spawn moves behind the report (the manager's matching handles this once
|
||||
the source flips); the `input` scenario proves the keyboard still types.
|
||||
|
||||
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
||||
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
||||
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
||||
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
||||
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
||||
has the shape of a signal every driver must answer, and it has no consumer
|
||||
until laptop sleep); CPU P/C-states.
|
||||
|
||||
## M21 preview — ACPI events + system power (planned next, not in this loop)
|
||||
|
||||
The acpi service grows the event side (settled direction 2026-07-13; detailed
|
||||
phases when M20 lands):
|
||||
|
||||
- **21.1 SCI + fixed events**: irq_bind the SCI (the resource M20.1 already
|
||||
records), read/clear PM1 status, publish the power-button event to
|
||||
subscribers (the same pub/sub shape the manager uses).
|
||||
- **21.2 GPE + Notify**: Notify dispatch in the shared AML interpreter, GPE
|
||||
block handling, `Notify(device, code)` published per reported node. The
|
||||
acpi service is a **bus** here: battery (PNP0C0A), AC (ACPI0003), and lid
|
||||
(PNP0C0D) nodes are reported children; small class drivers bind them and
|
||||
speak an evaluate/subscribe protocol to the service — the xHCI split,
|
||||
repeated. The embedded controller (`_Qxx` queries) rides this phase;
|
||||
QEMU emulates no battery/EC, so those paths are interface-complete and
|
||||
validated on real hardware (the laptop is the win condition), while the
|
||||
plumbing is proven by the power button.
|
||||
- **21.3 the capstone**: QEMU `system_powerdown` → acpi service event → init
|
||||
runs the M17 stop sequence over its children → kernel `\_S5` — orderly
|
||||
shutdown as the scenario that proves lifecycle + events compose. (The
|
||||
harness grows a QMP poke to inject the event.)
|
||||
+128
@@ -0,0 +1,128 @@
|
||||
# The power service: events and shutdown
|
||||
|
||||
A laptop lid closes, a battery drains, someone presses the power button — and
|
||||
several parts of the system might care: a session manager dims the screen, a
|
||||
logger notes it, and ultimately *something* has to turn the machine off. None of
|
||||
them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
Where the events come from is firmware-specific — on x86 they ride the ACPI SCI
|
||||
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
This is why the protocol is `power`, not "ACPI events": naming a cross-firmware
|
||||
surface after one firmware would leak x86 into code the ARM port must reuse
|
||||
unchanged.
|
||||
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
|
||||
| Direction | Operation | Purpose |
|
||||
|---|---|---|
|
||||
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||
|
||||
Events are published, not polled: like the input service, the service holds
|
||||
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||
message, so a slow or dead subscriber can never wedge the source. The event
|
||||
vocabulary is hardware-neutral:
|
||||
|
||||
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||
- `notify` — a device notification that maps to none of the above; its `code`
|
||||
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||
and what happened.
|
||||
|
||||
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||
generic `notify` is fully described without a second round trip.
|
||||
|
||||
**`shutdown` is authority, not information.** It is the only operation that
|
||||
*does* something irreversible, so it is gated: the contract is that only init
|
||||
(PID 1) may request it, because init is the process that has already run the stop
|
||||
sequence over everything else. The acpi service implements this as a **soft
|
||||
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||
init is the one subscriber. That stands in for "only the system supervisor may
|
||||
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||
is not init.
|
||||
|
||||
## Orderly shutdown
|
||||
|
||||
Powering off cleanly is where the power service, the [process
|
||||
lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init
|
||||
already supervises the services it starts; for shutdown it runs **one event loop
|
||||
over one endpoint** that carries three things at once: its children's exit
|
||||
notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||
**power events** it subscribes to — plus a re-arming heartbeat timer proving PID
|
||||
1 is alive. (init subscribes with retries, because the power service registers
|
||||
`.power` well after init starts; a missing power service is not fatal — a
|
||||
`terminate` signal drives the same path.)
|
||||
|
||||
On a `power_button` event or a `terminate` signal, init:
|
||||
|
||||
1. logs that it is shutting down,
|
||||
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||
last (other services may flush through it), each child getting the
|
||||
*terminate → deadline → kill* escalation from
|
||||
[process-lifecycle.md](process-lifecycle.md), and
|
||||
3. requests `.power` `shutdown`.
|
||||
|
||||
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||
the machine off, it logs loudly so a test fails rather than hangs.
|
||||
|
||||
**No new system call was needed for S5.** The broad io_port grant on the
|
||||
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||
could physically already do; formalizing it as a protocol operation added a
|
||||
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||
panic-time poweroff, where no user space is available to ask.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Two QEMU scenarios exercise the path, both injecting a real ACPI power-button
|
||||
press via QMP `system_powerdown` (there is no other deterministic power event on
|
||||
this config):
|
||||
|
||||
- `power-button` proves the source: the acpi service's SCI handler logs the
|
||||
press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)).
|
||||
- `orderly-shutdown` proves the whole composition: button → init logs shutting
|
||||
down → children stopped → the service enters S5 → QEMU exits. The ordered
|
||||
regex is the proof, and QEMU's self-exit through S5 is the pass.
|
||||
|
||||
## Scope
|
||||
|
||||
Interface-complete but validated on real hardware (the author's laptop) later,
|
||||
because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the
|
||||
interface stubs, lid and AC events, and the embedded controller's `_Qxx`
|
||||
queries. Deliberately out of scope for now: reboot over the power protocol, S3
|
||||
sleep, per-device D-states (a future lifecycle-vocabulary extension, since
|
||||
"suspend" has the shape of a signal every driver must answer and has no consumer
|
||||
until laptop sleep), and thermal zones.
|
||||
|
||||
## See also
|
||||
|
||||
- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power
|
||||
button fixed event, and GPE/Notify dispatch in the acpi service.
|
||||
- [discovery.md](discovery.md) — why the surface is domain-named, and the
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
@@ -17,12 +17,12 @@ was a mistake) without inheriting the mechanism, the API, or the names. The nami
|
||||
rule is danos's own and it is strict: plain words that communicate intent
|
||||
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: a
|
||||
**musl-based C layer** (growing out of library/posix) that wires C programs to the
|
||||
danos runtime — musl's syscall surface retargeted at danos system calls and IPC
|
||||
protocols (files onto the VFS protocol, `sigaction`/`wait` onto this lifecycle,
|
||||
sockets onto whatever networking becomes). Ported programs see POSIX; the system
|
||||
underneath never does.
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||
|
||||
## Why a standard vocabulary
|
||||
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||
more of the system moved into restartable processes (the discovery migration,
|
||||
[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing:
|
||||
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
|
||||
@@ -0,0 +1,227 @@
|
||||
# System Requirements
|
||||
|
||||
Minimum and recommended hardware for running danos. Every requirement below is
|
||||
grounded in what the current code actually assumes at boot — this is a
|
||||
description of the real target, not an aspirational one.
|
||||
|
||||
## Summary
|
||||
|
||||
danos targets a **modern UEFI x86-64 PC with ACPI and PCIe**. The practical
|
||||
minimum is:
|
||||
|
||||
- 64-bit x86-64 CPU with SSE2, APIC, and `syscall`/`sysret`
|
||||
- UEFI firmware (no BIOS / legacy boot)
|
||||
- ACPI tables: MADT, MCFG, FADT
|
||||
- PCIe with an ECAM (MMConfig) window
|
||||
- **128 MiB RAM** (target); see [Memory](#memory) for the breakdown
|
||||
- USB via **xHCI only**
|
||||
|
||||
There is no support for legacy BIOS boot, x2APIC, port-IO PCI configuration, or
|
||||
any USB host controller other than xHCI.
|
||||
|
||||
## Plain-language hardware guide
|
||||
|
||||
If you don't want to cross-reference chipset datasheets, here's roughly what era
|
||||
of PC works. These are **guidance based on when the required features became
|
||||
standard**, not a list of tested machines — the authoritative rules are in the
|
||||
technical sections below.
|
||||
|
||||
The feature that sets the floor is **built-in xHCI USB** (danos supports no other
|
||||
USB controller) combined with **UEFI firmware**. Both became standard on
|
||||
mainstream desktops and laptops around **2012**.
|
||||
|
||||
| | Known-good baseline | Comfortable recommendation |
|
||||
|---|---|---|
|
||||
| **Intel** | 3rd-gen Core "Ivy Bridge" (2012) with a 7-series "Panther Point" chipset — Intel's first chipset with xHCI built in | 6th-gen Core "Skylake" (2015) or newer |
|
||||
| **AMD** | A-series "Llano" APU with an A75 FCH (2011) — the industry's first chipset with built-in xHCI | Any AM4 platform, i.e. Ryzen (2017) or newer |
|
||||
|
||||
**AMD is not behind Intel here — it was first.** AMD's A75 FCH shipped with
|
||||
native xHCI in April 2011, about a year *ahead* of Intel's 7-series (2012); AMD
|
||||
was the first vendor to earn USB-IF certification for chipset-level USB 3.0. The
|
||||
two "comfortable recommendation" dates differ only because they name convenient,
|
||||
long-supported product lines (Skylake, Ryzen) — not because of any USB
|
||||
capability gap. Every AMD desktop platform from the A75 FCH (2011) and FM2/AM3+
|
||||
era onward has built-in xHCI, and any of them qualifies as a baseline.
|
||||
|
||||
Older 64-bit machines (e.g. Intel Core 2, Nehalem, Sandy Bridge) meet the CPU
|
||||
requirements but typically **lack built-in xHCI and/or ship with BIOS instead of
|
||||
UEFI**, so they are not supported.
|
||||
|
||||
### Matching your CPU by name
|
||||
|
||||
If you know your chip's marketing name or codename, find it here. Everything from
|
||||
the **Supported** rows down works; the **Too old** row does not.
|
||||
|
||||
**Intel Core** (the "-lake"/"-bridge"/"-well" codenames):
|
||||
|
||||
| Status | Generation | Codename(s) | Year |
|
||||
|---|---|---|---|
|
||||
| Too old | 2nd gen | Sandy Bridge | 2011 |
|
||||
| Supported (baseline) | 3rd gen | Ivy Bridge | 2012 |
|
||||
| Supported | 4th–5th gen | Haswell, Broadwell | 2013–2014 |
|
||||
| **Recommended** | 6th–9th gen | **Skylake**, Kaby Lake, Coffee Lake | 2015–2018 |
|
||||
| Recommended | 10th–11th gen | Comet Lake, Ice Lake, Tiger Lake, Rocket Lake | 2019–2021 |
|
||||
| Recommended | 12th gen+ | Alder Lake, Raptor Lake | 2021–2023 |
|
||||
| Recommended | Core Ultra | Meteor Lake, Arrow Lake, Lunar Lake | 2023+ |
|
||||
|
||||
**AMD:**
|
||||
|
||||
| Status | Family | Codename(s) | Year |
|
||||
|---|---|---|---|
|
||||
| Supported (baseline) | A-series APU (A75/A85 FCH) | Llano, Trinity, Richland, Kaveri | 2011–2014 |
|
||||
| Supported | FX (AM3+) | Bulldozer, Piledriver | 2011–2012 |
|
||||
| **Recommended** | **Ryzen** 1000–5000 (AM4) | Summit/Pinnacle Ridge, Matisse, Vermeer (Zen–Zen 3) | 2017–2020 |
|
||||
| Recommended | Ryzen 7000+ (AM5) | Raphael, Granite Ridge (Zen 4 / Zen 5) | 2022+ |
|
||||
| Recommended | Threadripper / EPYC | Zen and later | 2017+ |
|
||||
|
||||
(These map generations to the era their platforms shipped built-in xHCI + UEFI;
|
||||
they are guidance, not a tested-hardware list.)
|
||||
|
||||
**Two caveats that matter regardless of CPU:**
|
||||
|
||||
- **Firmware must be UEFI.** Many 2011-era machines could do either UEFI or
|
||||
legacy BIOS — danos needs it set to UEFI. There is no BIOS boot path.
|
||||
- **Input is PS/2 only, for now.** danos does not yet support USB
|
||||
keyboards/mice. This is fine on most **laptops** (their built-in keyboards are
|
||||
wired to a PS/2-style i8042 controller) but means a **desktop with only USB
|
||||
ports** currently has no usable keyboard. USB HID input is planned.
|
||||
|
||||
Virtual machines are the easiest way to meet every requirement: QEMU (with OVMF/
|
||||
UEFI, a `qemu-xhci` controller, and the default Q35 machine type), or any
|
||||
hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
## CPU / architecture
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:285`, `boot/efi.zig:418` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:282`, `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:59`, `isr.s:169` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:62`, `apic.zig:414` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:279`, `apic.zig:84` |
|
||||
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:144` |
|
||||
|
||||
## Firmware / boot
|
||||
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build.zig:464`, `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
(`efi.zig:578`, `boot-handoff.zig:144`)
|
||||
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||
(`system/devices/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel`, `/system/services/init`, and
|
||||
`/boot/initial-ramdisk.img` off the FAT boot volume. The kernel can boot
|
||||
"kernel-only" without init or the ramdisk. (`efi.zig:14`, `efi.zig:66`)
|
||||
|
||||
## Interrupt controller
|
||||
|
||||
- **Local APIC + I/O APIC required.** I/O APIC base, GSI base, and MADT
|
||||
interrupt-source overrides come from ACPI. (`cpu.zig:365`, `apic.zig:119`)
|
||||
- **MSI supported** — edge-triggered, keyed by vector, no I/O APIC mask cycle.
|
||||
Vector window 33–46, timer on 32, spurious on 47. (`system/kernel/irq.zig:70`,
|
||||
`cpu.zig:397`)
|
||||
- The legacy 8259 PIC is remapped and masked **only if present** (MADT
|
||||
`PCAT_COMPAT`); it is not required. (`apic.zig:103`)
|
||||
|
||||
## PCI / PCIe
|
||||
|
||||
- **PCIe with ECAM (MMConfig) required.** The PCI bus driver maps the host
|
||||
bridge's ECAM window (1 MiB config space per bus) and computes config
|
||||
addresses directly. **There is no legacy CF8/CFC port-IO config path** — the
|
||||
driver bails if the bridge exposes no ECAM window. The ECAM base comes from
|
||||
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:41`, `acpi.zig:6`)
|
||||
|
||||
## USB
|
||||
|
||||
- **xHCI only.** The sole USB driver is `usb-xhci-bus`, and the device manager
|
||||
binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI / OHCI / EHCI exist only
|
||||
as report strings with no driver behind them — **USB 1.x/2.0-only controllers
|
||||
are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||
`system/services/device-manager/device-manager.zig:34`)
|
||||
- USB input (keyboard/mouse over HID) is future work; the current input stack is
|
||||
PS/2. See [Buses & devices](#buses--devices).
|
||||
|
||||
## Timers
|
||||
|
||||
Calibration prefers, in order: (1) CPUID leaf `0x15` TSC frequency, (2) HPET,
|
||||
(3) ACPI PM timer (3.579545 MHz, from FADT), (4) legacy PIT. Any one suffices —
|
||||
HPET/PM-timer/PIT are optional fallbacks when CPUID `0x15` is absent.
|
||||
(`apic.zig:180`)
|
||||
|
||||
- **TSC** — monotonic high-resolution clock.
|
||||
- **LAPIC timer** — scheduler heartbeat, periodic at `timer_hz = 1000 Hz`.
|
||||
(`parameters.zig:39`)
|
||||
|
||||
## Memory
|
||||
|
||||
**Target: 128 MiB RAM.** The system uses 4 KiB pages and a bitmap physical-frame
|
||||
allocator built from the firmware memory map. There is no hardcoded minimum-RAM
|
||||
constant — the allocator only panics if there is no usable region, or none large
|
||||
enough to hold its own bitmap. (`system/kernel/pmm.zig:13`, `pmm.zig:77`)
|
||||
|
||||
Where the budget goes:
|
||||
|
||||
| Consumer | Size | Source |
|
||||
|---|---|---|
|
||||
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:26` |
|
||||
| Kernel stack, per CPU | 16 KiB | `parameters.zig:26` |
|
||||
| IST stack, per CPU | 16 KiB | `parameters.zig:36` |
|
||||
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:32` |
|
||||
| Max concurrent tasks | 32 | `parameters.zig:23` |
|
||||
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:299` |
|
||||
|
||||
The 64 MiB heap cap plus kernel image, per-CPU stacks, task stacks, the frame
|
||||
bitmap, and DMA-contiguous allocations fit comfortably within 128 MiB on a
|
||||
single- or low-core-count machine. Very high core counts (toward the 128-CPU
|
||||
ceiling) add per-CPU stack overhead and push toward more RAM.
|
||||
|
||||
**Note on the 4 GiB physmap:** the loader identity-maps and physmaps the low
|
||||
4 GiB of address space with 2 MiB leaves. This is *virtual address* reach, not a
|
||||
RAM requirement — RAM above 4 GiB simply needs an extra mapping window and is not
|
||||
needed to boot. (`efi.zig:305`)
|
||||
|
||||
Virtual-memory layout (`boot-handoff.zig:47`):
|
||||
|
||||
| Region | Base |
|
||||
|---|---|
|
||||
| User space | `0x0000_7000_0000_0000` |
|
||||
| Kernel heap | `0xFFFF_8000_0000_0000` |
|
||||
| Physmap | `0xFFFF_8800_0000_0000` |
|
||||
| Kernel image | `0xFFFF_FFFF_8000_0000` |
|
||||
|
||||
## Buses & devices
|
||||
|
||||
Buses with real drivers today:
|
||||
|
||||
- **PCIe** via ECAM (`pci-bus`)
|
||||
- **xHCI USB** (`usb-xhci-bus`)
|
||||
- **PS/2** keyboard + mouse (`ps2-bus`) — the current input stack
|
||||
- **Serial UART** (16550/16450), configured from the ACPI SPCR table
|
||||
|
||||
**No storage driver exists yet.** AHCI / NVMe / IDE are named for reporting only;
|
||||
there is no block-device driver. Persistent storage is future work.
|
||||
|
||||
## IOMMU
|
||||
|
||||
**Detection only; enforcement deferred.** The ACPI DMAR table is parsed for the
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInfo`
|
||||
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
|
||||
## What is explicitly NOT supported
|
||||
|
||||
- Legacy BIOS / multiboot / limine boot
|
||||
- 32-bit x86
|
||||
- x2APIC
|
||||
- Legacy port-IO (CF8/CFC) PCI configuration
|
||||
- Non-xHCI USB (UHCI / OHCI / EHCI)
|
||||
- Machines without ACPI (no device discovery)
|
||||
- Persistent storage (no AHCI / NVMe / IDE driver yet)
|
||||
- USB HID input (PS/2 only for now)
|
||||
@@ -27,6 +27,15 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||
so a serial-enabled image is still safe on real hardware.)
|
||||
|
||||
## In-kernel test cases
|
||||
|
||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||
|
||||
+117
@@ -0,0 +1,117 @@
|
||||
# Timers and time
|
||||
|
||||
Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
- **Reading the clock** — *what time is it?* A read of a free-running counter.
|
||||
- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.*
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
|
||||
## Why time is a syscall, not a service
|
||||
|
||||
The tempting microkernel move is to put a timer *driver* in user space and have
|
||||
applications ask it for the time over IPC. For a **monotonic clock that is wrong** —
|
||||
reading `now()` should never cost an IPC round trip. The kernel is already holding the
|
||||
answer: it computes the current time every time it schedules, from the TSC, in a couple
|
||||
of instructions. Surfacing that as a system call is pure mechanism; routing it through a
|
||||
message to another process would be slower *and* redundant, and a device like the HPET
|
||||
(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`.
|
||||
|
||||
This is the same conclusion every serious system reaches: Linux and Zircon read the
|
||||
counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the
|
||||
cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall.
|
||||
|
||||
That "from the TSC" hides a portability question, because the TSC is only a valid clock
|
||||
when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*.
|
||||
danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
## The three system calls
|
||||
|
||||
Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)):
|
||||
|
||||
- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not
|
||||
wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a
|
||||
128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of
|
||||
resolution, and just an `rdtsc` plus a multiply.
|
||||
- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake
|
||||
deadline and the tick sweep wakes it (`scheduler.sleep`).
|
||||
- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a
|
||||
**timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
both riding the scheduler tick.
|
||||
|
||||
## `runtime.time` — the generic interface
|
||||
|
||||
Applications don't call the syscalls directly; they use `runtime.time`
|
||||
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
front door, not new mechanism.
|
||||
|
||||
```zig
|
||||
const time = @import("runtime").time;
|
||||
|
||||
const start = time.now(); // Instant — monotonic
|
||||
doWork();
|
||||
const took = start.elapsed(); // Duration
|
||||
time.sleep(time.Duration.fromMillis(5)); // block ~5 ms
|
||||
|
||||
// A deadline delivered as a notification, so a service keeps serving meanwhile:
|
||||
_ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||
```
|
||||
|
||||
- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/
|
||||
fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's
|
||||
millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and
|
||||
returns early. All arithmetic saturates rather than wraps.
|
||||
- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` —
|
||||
built for deadline loops (`while (!deadline.reached()) …`).
|
||||
- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is
|
||||
calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller
|
||||
that needs real time can treat 0 as "unavailable" rather than assume it advances).
|
||||
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||
|
||||
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
|
||||
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||
owns covers every current use.
|
||||
|
||||
## Verifying it
|
||||
|
||||
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
|
||||
```
|
||||
$ zig build test # includes library/runtime/time.zig
|
||||
```
|
||||
|
||||
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||
`Duration`, read `now()` again, and the second reading is later — the kernel's timer
|
||||
driving a ring-3 program with no service in between.
|
||||
@@ -0,0 +1,349 @@
|
||||
# Running Zig on danos: the self-hosting roadmap
|
||||
|
||||
A design note (not built yet) on the path to making danos a **real Zig target** — a
|
||||
target you can name (`-target x86_64-danos`) and, eventually, run the Zig compiler
|
||||
itself on. It is forward-looking, like [vision.md](vision.md): it sets a direction
|
||||
and the decisions that follow from it, so the code we write now bends toward it
|
||||
instead of away.
|
||||
|
||||
This note deliberately does **not** cover a text editor or terminal. Those are
|
||||
easier (single-process, I/O-bound) and fall out of the early phases here almost for
|
||||
free; the hard, shaping problem is the standard-library surface, so that is what
|
||||
this roadmap is about.
|
||||
|
||||
The analysis behind it was done against **Zig 0.16** (the pinned toolchain). Zig's
|
||||
standard library moves between releases — especially the parts described here — so
|
||||
treat upstream references as "the shape in 0.16.x," and expect to re-check them on a
|
||||
toolchain bump.
|
||||
|
||||
## The win condition
|
||||
|
||||
danos runs the Zig compiler when a bare
|
||||
|
||||
```
|
||||
zig build-exe hello.zig
|
||||
```
|
||||
|
||||
completes **on danos** and produces a runnable danos binary. Note the milestone is
|
||||
`build-exe`, not `zig build`: the `zig build` runner spawns child processes (the
|
||||
build steps), which needs a whole process-control surface danos does not have yet.
|
||||
A single `build-exe` needs none of that (see Phase 3). Reaching `build-exe` is
|
||||
"self-hosting"; reaching `zig build` is a later, separate lift.
|
||||
|
||||
### Non-goals
|
||||
|
||||
- **No Linux syscall/ABI emulation.** danos will not implement the Linux `syscall`
|
||||
interface so that stock `x86_64-linux` binaries run. That is a permanent
|
||||
compatibility treadmill and it inverts the microkernel design — explicitly out.
|
||||
- **No musl port yet.** A musl libc port is a reasonable *later* effort (it unlocks
|
||||
the C ecosystem), but it is not on the critical path to Zig-on-danos, and it is
|
||||
deferred. The roadmap below is arranged so the work still pays off if musl ever
|
||||
happens (see "The same surface, twice").
|
||||
- **Editor/terminal are out of scope for this note** (they are downstream of Phase 1).
|
||||
|
||||
**On FFI.** Foreign-function interop splits the same way as the doors below. Zig-level
|
||||
and C-ABI-*exposing* FFI (`extern`, `callconv(.c)`, C-ABI structs) work on a real target
|
||||
immediately — and the `std.os.danos` seam is C-ABI-shaped by construction, so it is
|
||||
FFI-friendly from the start. *Consuming* C libraries (`@cImport`, linking archives) is
|
||||
the part that needs a libc + headers, i.e. the deferred musl door. So an eventual FFI
|
||||
need reinforces keeping that door open; it does not change the plan.
|
||||
|
||||
## The realization that shapes everything: 0.16 gives us *one* seam
|
||||
|
||||
The instinct "to target Zig we'd have to reimplement all the `std` namespaces" was
|
||||
how older Zig worked. Zig 0.16 (post-"writergate") is far kinder:
|
||||
|
||||
- **`std.fs` is essentially gone.** It is now path helpers plus deprecated aliases;
|
||||
there is no `std.fs.File`, `std.fs.Dir`, or `std.fs.cwd()`. File and directory
|
||||
work goes through **`std.Io`** — a single runtime **vtable** (`Io.zig`) of
|
||||
function pointers handed to `main` as `std.process.Init.io`. `std.Io.File` and
|
||||
`std.Io.Dir` are thin forwarders to that vtable. `Io.zig` and the `fs` shim carry
|
||||
**zero** per-OS branches.
|
||||
- **`std.posix` is one generic body** parameterised over a single `system` module.
|
||||
With no libc, `system` resolves **per target OS**: `.linux => std.os.linux`,
|
||||
`.plan9 => std.os.plan9`, and so on. The generic `std.posix.read`/`write`/`open`
|
||||
bodies are just `system.read(...)` plus an errno switch — *identical for every
|
||||
OS*. The only variable is what `system` binds to.
|
||||
- **`std.os.<tag>`** (e.g. `std/os/linux.zig`) is therefore the real porting seam: a
|
||||
low-level, C-ABI-shaped module of `read/write/open/close/lseek/mmap/clock/exit/…`
|
||||
plus an `errno` enum and the constant tables (`O_*`, `CLOCK_*`, `S_*`).
|
||||
|
||||
Put together: **to port danos we write `std.os.danos` once** — the ~30-operation
|
||||
seam — and the whole `std.posix` / `std.fs` / `std.Io` tower above it lights up
|
||||
generically, because none of it branches on the OS. That is a dramatically smaller
|
||||
and more contained target than "reimplement the namespaces."
|
||||
|
||||
## Three doors, and why we take the first
|
||||
|
||||
| Door | What it is | Verdict |
|
||||
|------|-----------|---------|
|
||||
| **1. Implement the std seam** (`std.os.danos`) | Write the ~30-op `system` module over danos's native ABI + VFS; the generic std tower lights up. | **Take this.** The only door that touches neither C nor the Linux ABI. |
|
||||
| **2. Port musl** | Port musl libc to danos, link Zig against it. | Defer. Good later for the *C* ecosystem; barely helps *Zig* (std only uses libc on the libc-linked path). |
|
||||
| **3. Emulate the Linux ABI** | Implement Linux syscalls so stock linux binaries run. | Reject. Bottomless compatibility treadmill; against the design. |
|
||||
|
||||
### The same surface, twice
|
||||
|
||||
Doors 1 and 2 are the **same native surface at different layers**. `std.posix.read`
|
||||
is `system.read(...)` + an errno switch *regardless of OS* — the only question is
|
||||
whether `system` is **`std.os.danos` (Zig)** or **musl (C)**. Either way, the set of
|
||||
danos-facing operations you must implement is the *same* ~30 ops, all bottoming out
|
||||
in danos's native syscalls + the VFS/FAT server.
|
||||
|
||||
So the runtime work below is **not throwaway** if musl ever happens: you are building
|
||||
the danos-native implementations of that surface either way. Door 1 just packages
|
||||
them as Zig; a future musl re-uses the identical kernel/VFS operations underneath. The
|
||||
two symmetries worth keeping in mind: doors 1 and 2 converge at the **top** (identical
|
||||
POSIX surface); doors 2 and 3 converge at the **bottom** (unmodified musl needs the
|
||||
Linux syscall ABI). Door 1 is the only one that avoids both C and Linux.
|
||||
|
||||
### A fork is table stakes — for any door
|
||||
|
||||
`std.Target.Os.Tag` is a **closed enum** baked into the compiler binary *and* into
|
||||
the `std` linked with every program; `-target x86_64-danos` resolves through it. So
|
||||
adding `danos` as a name requires patching and rebuilding the compiler — even the
|
||||
musl door needs this. "Fork Zig" is therefore not an extra cost unique to door 1; it
|
||||
is the price of admission for *any* real target. What door 1 adds on top is small and
|
||||
localised (below).
|
||||
|
||||
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||
|
||||
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||
|
||||
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||
(`read/write/open/close/lseek/mmap/munmap/clock/exit/…`) + an errno enum + the
|
||||
constant tables, each backed by danos's native syscalls and the VFS. **Structure it
|
||||
to mirror `std/os/linux.zig`.** This is the load-bearing, *non-throwaway* artifact:
|
||||
when we fork Zig, `runtime.os` is copy-pasted (near-verbatim) into `std.os.danos`.
|
||||
- **`runtime.fs` — the thin native file API** danos programs use *today*, layered
|
||||
over `runtime.os`. It is also the concrete backing for the `std.Io` vtable's
|
||||
file-write entry once we're a real target, which is why program stdout, diagnostics,
|
||||
and file writes should all be *decided once at that seam* rather than as bespoke
|
||||
per-call helpers (see "How this informs decisions now").
|
||||
|
||||
**Do not hand-mirror the high-level std namespaces.** `std.fs`/`std.Io`/`std.process`
|
||||
are generic and OS-agnostic; once `std.os.danos` exists and we fork, upstream *gives*
|
||||
them to danos for free. Hand-writing `runtime.std.fs` to imitate them would be
|
||||
redundant the day the fork works, and it would chase a moving target (0.16's `std.Io`
|
||||
is large and still shifting). Build the seam well; take the tower for free.
|
||||
|
||||
**Why not a library called `std`?** Because `@import("std")` resolves to the
|
||||
compiler-provided standard library; a user module named `std` would *shadow* it for
|
||||
anything that imports it that way. That is the real reason the seam lives *inside* a
|
||||
forked std as `std/os/danos.zig`, not as a `runtime.std` library — and why danos's end
|
||||
state (`@import("std")` just working, and knowing danos) is the most natively Zig it can
|
||||
be. `runtime.os` is only the interim staging ground: developed against the stock
|
||||
toolchain so Phase 1 need not wait on the fork, then promoted near-verbatim into the
|
||||
fork's `std/os/danos.zig`.
|
||||
|
||||
### Retire `library/posix`
|
||||
|
||||
The `posix` compatibility layer (`unistd`, `stdio`) was the right instinct too early.
|
||||
Its whole value is POSIX *spellings* for POSIX software — and danos has no POSIX
|
||||
software; every current caller is danos-native code that could use `runtime.fs`
|
||||
directly. The real POSIX story arrives later and from elsewhere (musl, or upstream
|
||||
`std`'s own posix over `std.os.danos`), which supersedes a hand-rolled shim. So it is
|
||||
premature abstraction that adds a "which layer do I use?" fork with no payoff yet.
|
||||
|
||||
Its footprint is tiny: **five** call sites, all `unistd` file operations —
|
||||
`system/services/fat/fat.zig` (`mount`), the `vfs-test` and `fat-test` clients, and
|
||||
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` is dead — nothing
|
||||
imports it. The plan: build `runtime.fs`, migrate those five to it, delete
|
||||
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary`.
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
What the seam needs, and what danos already provides:
|
||||
|
||||
| std need | danos today | Gap |
|
||||
|----------|-------------|-----|
|
||||
| open / read / write / close / lseek | VFS (via the current `unistd`, → `runtime.fs`) | none — repackage |
|
||||
| directory read (`getdents`) | VFS `readdir` | none — repackage |
|
||||
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||
| monotonic clock | `clock` syscall | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||
| wall-clock / realtime | done — `wall_clock` syscall (CMOS RTC, Phase 2d) | — |
|
||||
| **environment variables** | `Init` has no env field | missing (can start empty) |
|
||||
| **cwd / chdir** | paths are absolute or bare | missing (no cwd anchor) |
|
||||
| **entropy / random** | — | missing (needed behind `vtable.random`) |
|
||||
| process spawn + exit status | `system_spawn` starts a *named ramdisk binary*; `ExitReason` is a *category* | no exec-of-path, no numeric `WEXITSTATUS` |
|
||||
| threads | one thread per process | avoided via `-fsingle-threaded` (below) |
|
||||
| symlinks | `NodeKind` has the tag; unimplemented | low priority |
|
||||
|
||||
The clustering is clear: reads and memory are basically done; the real work is
|
||||
**filesystem mutation + richer stat + wall-clock**, and a few small seam pieces
|
||||
(page-allocator hook, stdio bytes, entropy). Process spawning and threads are
|
||||
side-stepped entirely for a single `build-exe`.
|
||||
|
||||
## The roadmap
|
||||
|
||||
### Phase 0 — Make `danos` a real target
|
||||
|
||||
**Host, target, self-host — keep the three roles straight.** The *host* is where the
|
||||
compiler runs (your mac + linux dev machines); the *target* is what it emits (`danos`);
|
||||
and eventually danos becomes a host too (self-hosting — the win condition). So the move
|
||||
is: fork the compiler, build it **for** your dev hosts, and teach it to **cross-compile
|
||||
to** danos. You already do this — danos is cross-compiled `freestanding` from your dev
|
||||
host today; Phase 0 swaps that `freestanding` target for a real `x86_64-danos` one, which
|
||||
is what unlocks the native `std`.
|
||||
|
||||
**Why a compiler fork, not just a `--zig-lib-dir` override.** `std.Target.Os.Tag` is a
|
||||
*closed enum compiled into the compiler binary*, so `-target x86_64-danos` will not even
|
||||
parse unless the compiler itself knows the tag. Overriding the std lib directory alone
|
||||
cannot add a target — and there is no libc-only shortcut (a future musl needs the same
|
||||
patch). The only alternative, staying on `freestanding` + hand-shims, is exactly the
|
||||
non-native feel we are leaving: `@import("std")` there is stubbed, not real.
|
||||
|
||||
**The fork.** Clone `ziglang/zig` at the pinned 0.16 tag; build it with a stock
|
||||
same-version `zig` (`zig build` in the tree — a standard, LLVM-pulling, roughly one-time
|
||||
build); point danos's `build.zig`/CI at the resulting binary. Four localised patches:
|
||||
|
||||
- add `danos` to `std.Target.Os.Tag`, in the "no version range" group alongside
|
||||
plan9/serenity;
|
||||
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||
not a prerequisite for starting).
|
||||
|
||||
This is the fork treadmill we accept once. Keep the patch set tiny and `else`-friendly,
|
||||
pin to one 0.16.x, and rebase on point releases.
|
||||
|
||||
### Phase 1 — `runtime.os` read-side + allocator + stdio + cwd; retire `posix`
|
||||
|
||||
Author `runtime.os` (→ `std.os.danos`): the `errno` enum, the constant tables, and
|
||||
the C-convention `read / write / open / openat / close / lseek / mmap / munmap /
|
||||
exit`, each returning result-or-`-errno`. Most backing already exists (VFS + native
|
||||
mmap + clock).
|
||||
|
||||
- Provide `page_allocator` via `root.os.heap.page_allocator` (a thin override over
|
||||
danos `mmap`). This sits **outside** the `std.Io` vtable, so it is wired separately.
|
||||
- Wire fd 0/1/2 to a console **byte** stream (today output only reaches `debug_write`;
|
||||
input is structured `InputEvent` IPC — a byte tty is a new, small thing in both
|
||||
directions).
|
||||
- Add a `getcwd`/`chdir` anchor so `std.fs.cwd()`-style resolution has something to
|
||||
resolve against.
|
||||
- Build `runtime.fs` over `runtime.os`; migrate the five `posix` callers to it; delete
|
||||
`library/posix/` and drop its build module.
|
||||
|
||||
After Phase 1, the surface an editor or terminal needs (open/read/write/close/lseek/
|
||||
readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||
|
||||
### Phase 2 — Filesystem mutation + real stat (the compiler's cache tower)
|
||||
|
||||
danos's biggest genuine gap, and the correctness-critical one:
|
||||
|
||||
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||
([protocol.zig](../system/services/vfs/protocol.zig)) and the FAT engine
|
||||
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||
(danos is monotonic-only today; an RTC/time service is the dependency).
|
||||
|
||||
Because `std.fs`/`std.Io` have no per-OS branches, finishing this in `runtime.os`
|
||||
lights up the whole file tower for the compiler at once. Environment can stay an empty
|
||||
map until the kernel populates a non-empty `envp`.
|
||||
|
||||
**Status — Phase 2 complete.** `truncate` (O_TRUNC, closing the boot-log stale-tail
|
||||
bug), `mkdir`, `unlink`, and `rename` are all wired through the FAT engine, the VFS
|
||||
protocol + router, and `runtime.fs` (`makeDirectory` / `remove` / `rename`) —
|
||||
host-tested and QEMU-tested (`fat-mutations` + `fat-rename` make a directory, write+read
|
||||
a file in it, rename it, then remove it through the mount). `removeFile` and `rename`
|
||||
are LFN-aware; `rename` is same-directory + 8.3 (cross-directory and long-name-
|
||||
preserving rename are noted limitations). Wall-clock is now a kernel syscall
|
||||
(`wall_clock`, a CMOS-RTC read anchored to the monotonic clock), and the FAT engine
|
||||
stamps and reports **mtime** — `stat` / `runtime.fs.Attributes` carry a real
|
||||
modification time (the `fat-mtime` case reads it back within seconds of the host clock).
|
||||
The remaining `stat` fields, `mode`/`inode`, are deferred (not needed until the
|
||||
compiler's cache layer wants them). **Everything past here is gated on Phase 0 (the
|
||||
fork):** the `runtime.os` seam, `cwd`, stdio-as-fds, and the compiler bring-up.
|
||||
|
||||
### Phase 3 — Single-threaded, self-linked compiler bring-up
|
||||
|
||||
Build the compiler with **two load-bearing flags**:
|
||||
|
||||
- **`-fsingle-threaded`** removes `std.Thread` entirely — `Thread.spawn` is a hard
|
||||
compile error under it, and `std.Io`'s threaded backend runs inline. danos being
|
||||
one-thread-per-process is therefore **not** a blocker. Parallel codegen is a
|
||||
throughput optimisation, not a correctness requirement.
|
||||
- **`-fno-llvm -fno-lld`** keeps codegen and linking **in-process** (the self-hosted
|
||||
x86-64 backend + self-linker), so a single `build-exe` **never forks a child**. That
|
||||
is what lets us defer the entire spawn/exec/wait surface.
|
||||
|
||||
Then supply the few remaining seam pieces: `now` (wrap the danos clock), an entropy
|
||||
source behind `vtable.random` (`randomSecure` can alias it initially — low volume, for
|
||||
temp-file names and hashmap seeds), and the Phase-2 mkdir/rename/unlink for cache dir
|
||||
trees and atomic temp-then-rename output.
|
||||
|
||||
**Explicitly deferred** (not on the `build-exe` path): child-process spawn/exec (only
|
||||
`zig build` and external tools need it), `std.Thread`, `fsync` (FAT is write-through
|
||||
today), symlinks, and musl.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **The std-fork rebase treadmill is the main ongoing cost.** A new OS tag touches the
|
||||
same broad file set plan9/serenity touch (hundreds of `native_os` sites, plus
|
||||
"unsupported OS" `@compileError` dead-ends a new tag must be routed around), and the
|
||||
entire `std.Io` layer is new in 0.16 and still moving. Stay pinned to one 0.16.x,
|
||||
keep additions localised and `else`-friendly. Watch the closed-enum gotcha: adding
|
||||
`danos` to `Os.Tag` can break existing *exhaustive* switches that lack an `else`, so
|
||||
expect to touch switch sites beyond the ones you implement.
|
||||
- **Single-threaded is load-bearing.** The "no `std.Thread`" simplification rests
|
||||
entirely on `-fsingle-threaded`. If a dependency or flag flips threading back on, you
|
||||
inherit an unescapable compile error (no root-hook exists) — the only outs are a full
|
||||
thread-impl fork or linking libc for pthreads. Keep `single_threaded` asserted end to
|
||||
end.
|
||||
- **In-process linking is load-bearing.** Reaching the compiler without fork/exec
|
||||
depends on `-fno-llvm -fno-lld`. The moment you shell out to LLD/`ld`, you need the
|
||||
full `spawn`/`wait` surface — the hardest microkernel piece — and danos's
|
||||
`system_spawn` only starts a *named ramdisk binary*, not exec of an arbitrary path.
|
||||
Verify the self-hosted backend covers the target output before assuming child
|
||||
processes are optional.
|
||||
- **The shim cannot host the compiler.** danos's current `runtime`/`posix` is fine for
|
||||
danos's *own* native programs, but the compiler `import`s *upstream* `std`, which on
|
||||
a non-target hits the void `system` stub. So the compiler forces the real target
|
||||
(Phase 0's fork). Do not over-invest in extending the hand-shim for compiler
|
||||
purposes; put that effort into `runtime.os` + the VFS/FAT operations, which both the
|
||||
fork *and* a future musl consume.
|
||||
- **`"w"`/`O_CREAT` does not truncate — a silent-corruption bug on this road.** The FAT
|
||||
engine's `writeFile` only *grows* `node.size`, so overwriting a shorter file leaves
|
||||
trailing garbage. Harmless for the boot log today, but for a compiler it means
|
||||
**corrupt `.o`/cache files that look like nondeterministic compiler bugs.** Land
|
||||
`truncate` (Phase 2) before the compiler ever writes cache.
|
||||
- **Exit status is categorical, not numeric.** `process_exit_reason` returns an
|
||||
`ExitReason` *category*, not a numeric code (`WEXITSTATUS`). Fine while spawn is
|
||||
stubbed; the day `zig build` or external tools arrive, plan a kernel exit-record
|
||||
extension — do not let it surprise you.
|
||||
|
||||
## How this informs decisions now
|
||||
|
||||
Two current decisions fall out of this roadmap:
|
||||
|
||||
1. **The `runtime.fs` / `std.Io` question resolves at the vtable seam.** Because 0.16
|
||||
routes *all* output through the `std.Io` vtable's file-write entry, and stdout/stderr
|
||||
are just `File`s with well-known handles, build `runtime.fs` (and the console stdout)
|
||||
as the concrete backing for that entry — not as a bespoke `std.Io.Writer`-only shim.
|
||||
Decide it once, at the seam, and program stdout, diagnostics, and file writes all
|
||||
flow through the same danos VFS/console path.
|
||||
2. **The boot-log `truncate` caveat is now fixed** (Phase 2a). It was the same
|
||||
`writeFile`-only-grows gap that on the self-hosting road would corrupt build output;
|
||||
`engine.truncate` + an O_TRUNC open flag now free the old chain so a shorter rewrite
|
||||
leaves no stale tail, and the boot-log flush opens with it.
|
||||
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the north star this serves.
|
||||
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
are confined, and now retired).
|
||||
@@ -1,13 +0,0 @@
|
||||
//! DanOS's POSIX / C compatibility layer — `unistd`, `stdio`, and (later) the C
|
||||
//! `errno` / `struct stat` / `extern "C"` surface. This is the *one* place POSIX and
|
||||
//! C spellings are allowed to appear verbatim (see docs/coding-standards.md): a file
|
||||
//! under library/posix/ *is* the foreign ABI, so it keeps the ABI's names. Everything
|
||||
//! it touches on the danos side (the VFS protocol, the runtime) uses danos names,
|
||||
//! which this layer translates to at the boundary.
|
||||
//!
|
||||
//! It is layered strictly *over* the runtime: it calls the runtime's IPC and heap,
|
||||
//! never the kernel's system calls directly. danos-native applications use the
|
||||
//! runtime; this exists so *POSIX* software can too.
|
||||
|
||||
pub const unistd = @import("unistd.zig");
|
||||
pub const stdio = @import("stdio.zig");
|
||||
@@ -1,115 +0,0 @@
|
||||
//! A small C stdio layer over the POSIX-style file API (unistd.zig). Unbuffered
|
||||
//! for now — each fread/fwrite is one VFS round trip; an internal buffer (fewer
|
||||
//! IPC calls) is a later optimisation. Both a Zig-callable API and `extern "C"`
|
||||
//! symbols are provided, so Zig and future C programs share it.
|
||||
|
||||
const std = @import("std");
|
||||
const unistd = @import("unistd.zig");
|
||||
const heap = @import("runtime").heap;
|
||||
|
||||
pub const SEEK_SET = unistd.SEEK_SET;
|
||||
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||
pub const SEEK_END = unistd.SEEK_END;
|
||||
|
||||
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
||||
/// heap; `fclose` frees it.
|
||||
pub const FILE = extern struct {
|
||||
fd: i32,
|
||||
eof: c_int = 0,
|
||||
err: c_int = 0,
|
||||
};
|
||||
|
||||
fn flagsFor(mode: []const u8) u32 {
|
||||
if (mode.len == 0) return 0;
|
||||
return switch (mode[0]) {
|
||||
'w', 'a' => unistd.O_CREAT,
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Open `path` in `mode` ("r"/"w"/"a", '+' ignored for now). Returns null on error.
|
||||
pub fn fopen(path: []const u8, mode: []const u8) ?*FILE {
|
||||
const fd = unistd.open(path, flagsFor(mode));
|
||||
if (fd < 0) return null;
|
||||
const f = heap.allocator().create(FILE) catch {
|
||||
unistd.close(fd);
|
||||
return null;
|
||||
};
|
||||
f.* = .{ .fd = fd };
|
||||
if (mode.len > 0 and mode[0] == 'a') _ = unistd.lseek(fd, 0, unistd.SEEK_END);
|
||||
return f;
|
||||
}
|
||||
|
||||
pub fn fclose(f: *FILE) c_int {
|
||||
unistd.close(f.fd);
|
||||
heap.allocator().destroy(f);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
||||
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||
const total = size * nmemb;
|
||||
if (total == 0) return 0;
|
||||
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
||||
if (n <= 0) {
|
||||
f.eof = 1;
|
||||
return 0;
|
||||
}
|
||||
return @as(usize, @intCast(n)) / size;
|
||||
}
|
||||
|
||||
/// Write `size*nmemb` bytes; returns the number of whole items written.
|
||||
pub fn fwrite(data: []const u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||
const total = @min(data.len, size * nmemb);
|
||||
if (total == 0) return 0;
|
||||
const n = unistd.write(f.fd, data[0..total]);
|
||||
if (n <= 0) {
|
||||
f.err = 1;
|
||||
return 0;
|
||||
}
|
||||
return @as(usize, @intCast(n)) / size;
|
||||
}
|
||||
|
||||
pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
||||
f.eof = 0;
|
||||
return if (unistd.lseek(f.fd, off, whence) < 0) -1 else 0;
|
||||
}
|
||||
|
||||
pub fn ftell(f: *FILE) i64 {
|
||||
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
||||
}
|
||||
|
||||
pub fn rewind(f: *FILE) void {
|
||||
_ = fseek(f, 0, SEEK_SET);
|
||||
}
|
||||
|
||||
pub fn feof(f: *FILE) c_int {
|
||||
return f.eof;
|
||||
}
|
||||
|
||||
pub fn ferror(f: *FILE) c_int {
|
||||
return f.err;
|
||||
}
|
||||
|
||||
pub fn fputs(s: []const u8, f: *FILE) c_int {
|
||||
return if (unistd.write(f.fd, s) < 0) -1 else 0;
|
||||
}
|
||||
|
||||
pub fn fputc(c: u8, f: *FILE) c_int {
|
||||
const b = [_]u8{c};
|
||||
return if (unistd.write(f.fd, &b) == 1) c else -1;
|
||||
}
|
||||
|
||||
pub fn fgetc(f: *FILE) c_int {
|
||||
var b: [1]u8 = undefined;
|
||||
const n = unistd.read(f.fd, &b);
|
||||
if (n <= 0) {
|
||||
f.eof = 1;
|
||||
return -1; // EOF
|
||||
}
|
||||
return b[0];
|
||||
}
|
||||
|
||||
// Real `extern "C"` symbols (fopen/fread/fseek/...) — with a C-string signature
|
||||
// distinct from the Zig slice API above — land with the first C program, wired
|
||||
// via @export so they don't collide with these Zig names.
|
||||
@@ -1,148 +0,0 @@
|
||||
//! POSIX-style file API for user programs — the low level under C stdio. Files
|
||||
//! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
|
||||
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
|
||||
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
||||
|
||||
const std = @import("std");
|
||||
const protocol = @import("vfs-protocol");
|
||||
const ipc = @import("runtime").ipc;
|
||||
|
||||
pub const O_CREAT = protocol.create;
|
||||
pub const SEEK_SET: u32 = 0;
|
||||
pub const SEEK_CURRENT: u32 = 1;
|
||||
pub const SEEK_END: u32 = 2;
|
||||
|
||||
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
||||
var vfs_handle: usize = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?usize {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const maximum_fds = 32;
|
||||
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
||||
var fds = [_]Fd{.{}} ** maximum_fds;
|
||||
|
||||
fn allocFd() ?usize {
|
||||
for (&fds, 0..) |*f, i| {
|
||||
if (!f.used) {
|
||||
f.* = .{ .used = true };
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||
const rpl = @min(n - protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
||||
pub fn open(path: []const u8, flags: u32) i32 {
|
||||
const fd = allocFd() orelse return -1;
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
||||
const r = transact(request, path, &.{}) orelse {
|
||||
fds[fd].used = false;
|
||||
return -1;
|
||||
};
|
||||
if (r.reply.status != 0) {
|
||||
fds[fd].used = false;
|
||||
return -1;
|
||||
}
|
||||
fds[fd] = .{ .used = true, .node = r.reply.node, .offset = 0 };
|
||||
return @intCast(fd);
|
||||
}
|
||||
|
||||
fn fdPtr(fd: i32) ?*Fd {
|
||||
if (fd < 0 or fd >= maximum_fds) return null;
|
||||
const f = &fds[@intCast(fd)];
|
||||
return if (f.used) f else null;
|
||||
}
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
||||
pub fn read(fd: i32, buffer: []u8) isize {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return -1;
|
||||
if (r.reply.status != 0) return -1;
|
||||
f.offset += r.reply.len;
|
||||
return @intCast(r.reply.len);
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count or -1.
|
||||
pub fn write(fd: i32, data: []const u8) isize {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return -1;
|
||||
if (r.reply.status != 0) return -1;
|
||||
f.offset += r.reply.len;
|
||||
return @intCast(r.reply.len);
|
||||
}
|
||||
|
||||
/// Reposition the fd's offset. Returns the new offset or -1. (SEEK_END needs the
|
||||
/// file size, which `stat` provides; handled by fetching it here.)
|
||||
pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const base: i64 = switch (whence) {
|
||||
SEEK_SET => 0,
|
||||
SEEK_CURRENT => @intCast(f.offset),
|
||||
SEEK_END => blk: {
|
||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||
const st = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||
break :blk @intCast(st.size);
|
||||
},
|
||||
else => return -1,
|
||||
};
|
||||
const pos = base + off;
|
||||
if (pos < 0) return -1;
|
||||
f.offset = @intCast(pos);
|
||||
return pos;
|
||||
}
|
||||
|
||||
/// Stat `path`. Returns 0 or -1.
|
||||
pub fn stat(path: []const u8, out: *protocol.FileStatus) i32 {
|
||||
// Open, stat by node, close — simple and enough for now.
|
||||
const fd = open(path, 0);
|
||||
if (fd < 0) return -1;
|
||||
defer close(fd);
|
||||
const f = fdPtr(fd).?;
|
||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||
out.* = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Close an fd (best effort — tells the VFS to release the open file).
|
||||
pub fn close(fd: i32) void {
|
||||
const f = fdPtr(fd) orelse return;
|
||||
const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
f.used = false;
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
//! Block-device client: the helper a filesystem uses to read and write a block
|
||||
//! device (a USB stick, via usb-storage) without hand-rolling the block-protocol
|
||||
//! IPC. Layered over `ipc` and the shared `block-protocol` wire format, like
|
||||
//! `runtime.usb` over the transfer protocol.
|
||||
//!
|
||||
//! Transfers name a caller-owned DMA buffer by physical address (from
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("block-protocol");
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
endpoint: ipc.Handle,
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (result.status != 0) return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.write, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// Look up the block device, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 1200) : (attempts += 1) {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,274 @@
|
||||
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||
//! lists files served by the user-space VFS (system/services/vfs), each call
|
||||
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||
//! danos programs use directly; it is also where the file operations that later
|
||||
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||
//! the old POSIX `unistd` shim — a compatibility spelling danos does not need yet.
|
||||
//!
|
||||
//! Handles are *values*, not entries in a global descriptor table: a `File` /
|
||||
//! `Directory` owns its VFS node id and (for files) a byte offset. So there is no
|
||||
//! per-process fd limit and no shared table to synchronise — the danos-native
|
||||
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
/// wire protocol.
|
||||
pub const Kind = protocol.NodeKind;
|
||||
|
||||
/// A node's metadata (the answer to a status request).
|
||||
pub const Attributes = struct {
|
||||
size: u64,
|
||||
kind: Kind,
|
||||
/// Modification time — Unix epoch seconds, UTC. 0 if the filesystem has none.
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
// Map a wire `NodeKind` value to the enum, defaulting anything unrecognised to
|
||||
// `.regular` (the server is trusted, but a value outside the enum would be
|
||||
// illegal to `@enumFromInt` directly).
|
||||
fn kindFromWire(value: u32) Kind {
|
||||
return switch (value) {
|
||||
@intFromEnum(Kind.directory) => .directory,
|
||||
@intFromEnum(Kind.character_device) => .character_device,
|
||||
@intFromEnum(Kind.block_device) => .block_device,
|
||||
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||
@intFromEnum(Kind.fifo) => .fifo,
|
||||
@intFromEnum(Kind.socket) => .socket,
|
||||
else => .regular,
|
||||
};
|
||||
}
|
||||
|
||||
/// How to open a path.
|
||||
pub const OpenOptions = struct {
|
||||
/// Create the file if it does not exist.
|
||||
create: bool = false,
|
||||
/// Open a directory node (for listing) rather than a file.
|
||||
directory: bool = false,
|
||||
/// Truncate an existing file to zero length on open (O_TRUNC) — replace its
|
||||
/// contents rather than overwriting in place.
|
||||
truncate: bool = false,
|
||||
|
||||
fn wireFlags(self: OpenOptions) u32 {
|
||||
var f: u32 = 0;
|
||||
if (self.create) f |= protocol.create;
|
||||
if (self.directory) f |= protocol.directory;
|
||||
if (self.truncate) f |= protocol.truncate;
|
||||
return f;
|
||||
}
|
||||
};
|
||||
|
||||
// The VFS server endpoint, looked up once by well-known id and cached.
|
||||
var vfs_handle: ipc.Handle = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?ipc.Handle {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||
const rpl = @min(n - protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||
pub const File = struct {
|
||||
node: u64,
|
||||
offset: u64 = 0,
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
/// call is capped at the VFS payload size, so the return may be short — use
|
||||
/// `writeAll` to write the whole slice. Null on error.
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||
/// total written, or null if a write failed before any progress.
|
||||
pub fn writeAll(self: *File, data: []const u8) ?usize {
|
||||
var written: usize = 0;
|
||||
while (written < data.len) {
|
||||
const n = self.write(data[written..]) orelse return if (written == 0) null else written;
|
||||
if (n == 0) return written; // no forward progress; stop rather than spin
|
||||
written += n;
|
||||
}
|
||||
return written;
|
||||
}
|
||||
|
||||
/// Move the read/write cursor to an absolute byte position.
|
||||
pub fn seekTo(self: *File, position: u64) void {
|
||||
self.offset = position;
|
||||
}
|
||||
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this file.
|
||||
pub fn close(self: *File) void {
|
||||
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
|
||||
const r = transact(request, path, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node };
|
||||
}
|
||||
|
||||
/// A path's metadata without keeping it open (open -> status -> close).
|
||||
pub fn attributes(path: []const u8) ?Attributes {
|
||||
var file = open(path, .{}) orelse return null;
|
||||
defer file.close();
|
||||
return file.attributes();
|
||||
}
|
||||
|
||||
/// Whether `path` resolves — handy as a readiness check (e.g. waiting for a mount
|
||||
/// to come up before writing to it).
|
||||
pub fn exists(path: []const u8) bool {
|
||||
return attributes(path) != null;
|
||||
}
|
||||
|
||||
/// One entry returned by `Directory.next`.
|
||||
pub const Entry = struct {
|
||||
kind: Kind = .regular,
|
||||
size: u64 = 0,
|
||||
name_buffer: [64]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
|
||||
pub fn name(self: *const Entry) []const u8 {
|
||||
return self.name_buffer[0..self.name_len];
|
||||
}
|
||||
};
|
||||
|
||||
/// An open directory being listed, cursor-advanced by `next`.
|
||||
pub const Directory = struct {
|
||||
node: u64,
|
||||
cursor: u64 = 0,
|
||||
|
||||
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||
entry.kind = kindFromWire(header.kind);
|
||||
entry.size = header.size;
|
||||
const source = r.payload[protocol.directory_entry_size..];
|
||||
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this directory.
|
||||
pub fn close(self: *Directory) void {
|
||||
var f = File{ .node = self.node };
|
||||
f.close();
|
||||
}
|
||||
};
|
||||
|
||||
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||
pub fn openDirectory(path: []const u8) ?Directory {
|
||||
const file = open(path, .{ .directory = true }) orelse return null;
|
||||
return .{ .node = file.node };
|
||||
}
|
||||
|
||||
// A path-based request that returns only a status (mkdir, unlink).
|
||||
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
|
||||
const r = transact(request, path, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||
/// success. Only works under a mounted filesystem that supports directories.
|
||||
pub fn makeDirectory(path: []const u8) bool {
|
||||
return pathOperation(.mkdir, path);
|
||||
}
|
||||
|
||||
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||
/// (a separate directory-removal would have to check emptiness).
|
||||
pub fn remove(path: []const u8) bool {
|
||||
return pathOperation(.unlink, path);
|
||||
}
|
||||
|
||||
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
|
||||
/// directory, 8.3-name rename only for now). Returns true on success.
|
||||
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const total = old_path.len + 1 + new_path.len;
|
||||
if (total > protocol.maximum_payload) return false;
|
||||
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_path.len], old_path);
|
||||
payload[old_path.len] = 0;
|
||||
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
/// the VFS then routes everything under `target` to that backend. This is the one
|
||||
/// call that hands the VFS a capability (the backend endpoint). Returns true on
|
||||
/// success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
const h = vfs() orelse return false;
|
||||
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const tlen = @min(target.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
|
||||
if (result.len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
@@ -12,6 +12,9 @@
|
||||
//! (arguments arrive via `init`).
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
@@ -20,6 +23,9 @@ pub const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||
pub const power_protocol = @import("power-protocol");
|
||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||
/// service. See library/runtime/input.zig and system/services/input/.
|
||||
pub const input = @import("input.zig");
|
||||
@@ -32,6 +38,19 @@ pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||
pub const usb = @import("usb.zig");
|
||||
|
||||
/// Block-device client: read/write a block device (a USB stick, via
|
||||
/// usb-storage). See library/runtime/block.zig.
|
||||
pub const block = @import("block.zig");
|
||||
|
||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||
/// layer danos programs use directly, and where the operations that later become
|
||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||
pub const fs = @import("fs.zig");
|
||||
|
||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||
pub const panic = start.panic;
|
||||
|
||||
|
||||
@@ -51,6 +51,24 @@ pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
|
||||
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
|
||||
/// filesystem stamps as a file's modification time. Formatting it into a calendar
|
||||
/// date/timezone is user-space policy layered on top.
|
||||
pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Copy bytes out of the kernel's in-memory diagnostic log — the accumulated
|
||||
/// stream of everything `write` (and the kernel itself) has emitted — starting at
|
||||
/// `offset`, into `out`. Returns the number of bytes copied (0 at end of buffer).
|
||||
/// A program reads the whole log by looping from offset 0, advancing by the return
|
||||
/// value, until it gets 0. This is how the boot log is persisted to disk on a
|
||||
/// headless/real machine where serial output is otherwise lost.
|
||||
pub fn klogRead(offset: usize, out: []u8) usize {
|
||||
return sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||
//!
|
||||
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||
//!
|
||||
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||
//! CLOCK_REALTIME) layered on top later.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
const nanos_per_micro: u64 = 1_000;
|
||||
const nanos_per_milli: u64 = 1_000_000;
|
||||
const nanos_per_second: u64 = 1_000_000_000;
|
||||
|
||||
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||
/// kernel's millisecond granularity and rounding down could return early.
|
||||
pub const Duration = struct {
|
||||
ns: u64,
|
||||
|
||||
pub fn fromNanos(n: u64) Duration {
|
||||
return .{ .ns = n };
|
||||
}
|
||||
pub fn fromMicros(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_micro };
|
||||
}
|
||||
pub fn fromMillis(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_milli };
|
||||
}
|
||||
pub fn fromSeconds(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_second };
|
||||
}
|
||||
|
||||
pub fn asNanos(d: Duration) u64 {
|
||||
return d.ns;
|
||||
}
|
||||
pub fn asMicros(d: Duration) u64 {
|
||||
return d.ns / nanos_per_micro;
|
||||
}
|
||||
pub fn asMillis(d: Duration) u64 {
|
||||
return d.ns / nanos_per_milli;
|
||||
}
|
||||
pub fn asSeconds(d: Duration) u64 {
|
||||
return d.ns / nanos_per_second;
|
||||
}
|
||||
|
||||
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||
pub fn ceilMillis(d: Duration) u64 {
|
||||
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||
}
|
||||
|
||||
pub fn plus(a: Duration, b: Duration) Duration {
|
||||
return .{ .ns = a.ns +| b.ns };
|
||||
}
|
||||
};
|
||||
|
||||
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||
/// saturate at zero rather than wrap.
|
||||
pub const Instant = struct {
|
||||
ns: u64,
|
||||
|
||||
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||
return .{ .ns = self.ns -| earlier.ns };
|
||||
}
|
||||
|
||||
/// How long since this instant, sampled now.
|
||||
pub fn elapsed(self: Instant) Duration {
|
||||
return now().since(self);
|
||||
}
|
||||
|
||||
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||
pub fn plus(self: Instant, d: Duration) Instant {
|
||||
return .{ .ns = self.ns +| d.ns };
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||
pub fn reached(deadline: Instant) bool {
|
||||
return now().ns >= deadline.ns;
|
||||
}
|
||||
};
|
||||
|
||||
/// The current monotonic time.
|
||||
pub fn now() Instant {
|
||||
return .{ .ns = system.clock() };
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||
/// want a plain integer instead of an `Instant`.
|
||||
pub fn monotonicNanos() u64 {
|
||||
return system.clock();
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||
/// "unavailable" instead of assuming the clock advances.
|
||||
pub fn available() bool {
|
||||
return system.clock() != 0;
|
||||
}
|
||||
|
||||
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||
/// `spin`.
|
||||
pub fn sleep(d: Duration) void {
|
||||
system.sleep(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
system.sleep(ms);
|
||||
}
|
||||
|
||||
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||
/// Prefer `sleep` for anything at or above a millisecond.
|
||||
pub fn spin(d: Duration) void {
|
||||
const deadline = now().plus(d);
|
||||
while (!deadline.reached()) {}
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||
pub fn after(endpoint: usize, d: Duration) bool {
|
||||
return system.timerOnce(endpoint, d.ceilMillis());
|
||||
}
|
||||
|
||||
test "Duration unit conversions round toward zero" {
|
||||
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||
}
|
||||
|
||||
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||
}
|
||||
|
||||
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||
const t0 = Instant{ .ns = 1_000 };
|
||||
const t1 = Instant{ .ns = 4_000 };
|
||||
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||
}
|
||||
|
||||
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
//! USB class-driver client: the helper a keyboard, mouse, or mass-storage driver
|
||||
//! uses to reach its device through the xHCI bus driver, so it never hand-rolls
|
||||
//! the transfer-protocol IPC. Layered over `ipc` and the shared
|
||||
//! `usb-transfer-protocol` wire format, the way `input.zig` layers over the input
|
||||
//! service and `device.zig` over the raw device calls.
|
||||
//!
|
||||
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||
//! if (!usb.helloManager(id)) return; // meet the spawn deadline
|
||||
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||
//! while (true) { ... ipc.replyWait(device.endpoint, ...) ... } // its own loop
|
||||
//!
|
||||
//! Reports are delivered to `device.endpoint` as asynchronous `InterruptReport`
|
||||
//! messages (the class driver runs a bare `replyWait` loop to read them, because
|
||||
//! the service harness drops buffered-message payloads — see service.zig).
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("usb-transfer-protocol");
|
||||
const device_manager = @import("device-manager-protocol");
|
||||
|
||||
pub const Endpoint = protocol.Endpoint;
|
||||
pub const InterruptReport = protocol.InterruptReport;
|
||||
pub const max_report_data = protocol.max_report_data;
|
||||
|
||||
// Endpoint transfer types (EndpointDescriptor attributes), for `findEndpoint`.
|
||||
pub const transfer_type_bulk: u8 = 2;
|
||||
pub const transfer_type_interrupt: u8 = 3;
|
||||
|
||||
/// An opened USB device: the bus endpoint to send requests to, this driver's own
|
||||
/// endpoint that reports arrive on, the device token, and the interface's
|
||||
/// endpoints (so a driver need not re-read the configuration descriptor).
|
||||
pub const Device = struct {
|
||||
bus: ipc.Handle,
|
||||
endpoint: ipc.Handle,
|
||||
token: u64,
|
||||
class: u8,
|
||||
subclass: u8,
|
||||
protocol_code: u8,
|
||||
interface_number: u8,
|
||||
endpoint_count: usize = 0,
|
||||
endpoints: [protocol.max_reported_endpoints]Endpoint = undefined,
|
||||
|
||||
/// The interface's first endpoint of the given transfer type and direction
|
||||
/// (`transfer_type_bulk` / `transfer_type_interrupt`), or null.
|
||||
pub fn findEndpoint(self: *const Device, transfer_type: u8, direction_in: bool) ?Endpoint {
|
||||
for (self.endpoints[0..self.endpoint_count]) |endpoint| {
|
||||
if (endpoint.transfer_type == transfer_type and (endpoint.address & 0x80 != 0) == direction_in) return endpoint;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||
var request = protocol.ControlRequest{
|
||||
.device_token = self.token,
|
||||
.setup = setup,
|
||||
.direction_in = @intFromBool(direction_in),
|
||||
.data_length = @intCast(data.len),
|
||||
};
|
||||
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||
var reply: [@sizeOf(protocol.ControlReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (length < @sizeOf(protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(protocol.ControlReply, reply[0..@sizeOf(protocol.ControlReply)]);
|
||||
if (control_reply.status != 0) return null;
|
||||
const actual = @min(control_reply.actual_length, data.len);
|
||||
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||
return actual;
|
||||
}
|
||||
|
||||
/// A control transfer with no data stage (SET_PROTOCOL, SET_IDLE, ...). The
|
||||
/// `setup` is a bit-cast `usb_abi.Request`.
|
||||
pub fn controlOut(self: *Device, setup: [8]u8) bool {
|
||||
return self.controlTransfer(setup, false, &.{}) != null;
|
||||
}
|
||||
|
||||
/// A device-to-host control transfer, returning the bytes read into `out`.
|
||||
pub fn controlIn(self: *Device, setup: [8]u8, out: []u8) ?usize {
|
||||
return self.controlTransfer(setup, true, out);
|
||||
}
|
||||
|
||||
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||
var request = protocol.InterruptSubscribeRequest{
|
||||
.device_token = self.token,
|
||||
.endpoint_address = endpoint_address,
|
||||
.max_length = max_length,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (length < @sizeOf(protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(protocol.InterruptSubscribeReply, reply[0..@sizeOf(protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
var request = protocol.BulkRequest{
|
||||
.device_token = self.token,
|
||||
.physical_address = physical,
|
||||
.length = length,
|
||||
.endpoint_address = endpoint_address,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.BulkReply)]u8 = undefined;
|
||||
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (replied < @sizeOf(protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(protocol.BulkReply, reply[0..@sizeOf(protocol.BulkReply)]);
|
||||
if (bulk_reply.status != 0) return null;
|
||||
return bulk_reply.actual_length;
|
||||
}
|
||||
};
|
||||
|
||||
/// Look up the USB bus and open the device with the assigned id, handing over a
|
||||
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
|
||||
/// bus is still coming up (a class driver races the bus driver at boot).
|
||||
pub fn open(device_id: u64) ?Device {
|
||||
var attempts: usize = 0;
|
||||
const bus = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
} else return null;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
var request = protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(protocol.OpenReply)]u8 = undefined;
|
||||
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < @sizeOf(protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(protocol.OpenReply, reply[0..@sizeOf(protocol.OpenReply)]);
|
||||
if (open_reply.status != 0) return null;
|
||||
|
||||
var device = Device{
|
||||
.bus = bus,
|
||||
.endpoint = endpoint,
|
||||
.token = open_reply.device_token,
|
||||
.class = open_reply.interface_class,
|
||||
.subclass = open_reply.interface_subclass,
|
||||
.protocol_code = open_reply.interface_protocol,
|
||||
.interface_number = open_reply.interface_number,
|
||||
.endpoint_count = @min(open_reply.endpoint_count, protocol.max_reported_endpoints),
|
||||
};
|
||||
for (0..device.endpoint_count) |index| device.endpoints[index] = open_reply.endpoints[index];
|
||||
return device;
|
||||
}
|
||||
|
||||
/// Hello the device manager as a class driver (Role.device) so a supervised
|
||||
/// spawn meets its hello deadline. Retries while the manager comes up.
|
||||
pub fn helloManager(device_id: u64) bool {
|
||||
var attempts: usize = 0;
|
||||
const manager = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
} else return false;
|
||||
|
||||
const hello = device_manager.Hello{ .role = @intFromEnum(device_manager.Role.device), .device_id = device_id };
|
||||
var reply: [device_manager.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch return false;
|
||||
if (length < device_manager.reply_size) return false;
|
||||
return std.mem.bytesToValue(device_manager.HelloReply, reply[0..device_manager.reply_size]).status == 0;
|
||||
}
|
||||
@@ -58,6 +58,8 @@ pub const SystemCall = enum(u64) {
|
||||
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||
_,
|
||||
};
|
||||
|
||||
@@ -177,6 +179,10 @@ pub const ServiceId = enum(u32) {
|
||||
input = 2,
|
||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||
_,
|
||||
};
|
||||
|
||||
|
||||
+45
-437
@@ -3,11 +3,13 @@
|
||||
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||
//! bootloader handed us) and translates the static tables into the generic
|
||||
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
||||
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
||||
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
||||
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
||||
//! interpretation is a separate, larger subproject.
|
||||
//! source. This is deliberately the *static-table* path, and **only** that: MADT
|
||||
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
|
||||
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
|
||||
//! interpreted here — the kernel collects the blobs and publishes them on the
|
||||
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
|
||||
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
|
||||
//! off the single-core critical path (nothing else runs alongside it there).
|
||||
//!
|
||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||
@@ -17,10 +19,8 @@
|
||||
const std = @import("std");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const parameters = @import("parameters");
|
||||
const device_model = @import("device-model.zig");
|
||||
const aml = @import("aml/aml.zig");
|
||||
const DeviceTree = device_model.DeviceTree;
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
@@ -38,8 +38,11 @@ pub const RegisterAccess = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||
/// The power register map, extracted from the FADT during discovery. Populated by
|
||||
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
|
||||
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
|
||||
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
|
||||
/// writes the PM1 control register itself). So this holds only the FADT scalars.
|
||||
pub const PowerInformation = struct {
|
||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||
@@ -55,10 +58,6 @@ pub const PowerInformation = struct {
|
||||
reset: RegisterAccess = .{},
|
||||
reset_value: u8 = 0,
|
||||
reset_supported: bool = false,
|
||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
||||
/// packages.
|
||||
s5: ?aml.SleepType = null,
|
||||
s3: ?aml.SleepType = null,
|
||||
};
|
||||
|
||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||
@@ -142,25 +141,20 @@ const maximum_cpus = parameters.maximum_cpus;
|
||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||
pub var cpu_information: CpuInformation = .{};
|
||||
|
||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
||||
pub const AmlStats = struct {
|
||||
nodes: usize = 0,
|
||||
consumed: usize = 0,
|
||||
total: usize = 0,
|
||||
};
|
||||
pub var aml_stats: AmlStats = .{};
|
||||
|
||||
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
||||
/// device enumeration later. Null until `discover` runs successfully.
|
||||
pub var namespace: ?aml.Namespace = null;
|
||||
|
||||
/// Physical address of the DSDT the FADT points at, or 0.
|
||||
pub var dsdt_physical: u64 = 0;
|
||||
|
||||
/// The FADT itself (physical + length), published on the acpi-tables node so
|
||||
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
||||
/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML
|
||||
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
||||
var fadt_physical: u64 = 0;
|
||||
var fadt_length: u64 = 0;
|
||||
|
||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||
// for the sleep-state (`_Sx`) packages.
|
||||
// address + length of each table's post-header bytecode. The kernel does not
|
||||
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
|
||||
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
|
||||
var aml_block_physical: [32]u64 = undefined;
|
||||
var aml_block_len: [32]usize = undefined;
|
||||
var aml_block_count: usize = 0;
|
||||
@@ -367,16 +361,16 @@ const Hpet = extern struct {
|
||||
|
||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
|
||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
if (rsdp_physical == 0) return error.NoRsdp;
|
||||
boot_memory_regions = memory_regions;
|
||||
|
||||
// Start clean so a re-run doesn't accumulate stale state.
|
||||
power_information = .{};
|
||||
fadt_physical = 0;
|
||||
fadt_length = 0;
|
||||
platform_information = .{};
|
||||
aml_stats = .{};
|
||||
namespace = null;
|
||||
dsdt_physical = 0;
|
||||
aml_block_count = 0;
|
||||
|
||||
@@ -393,36 +387,20 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||
}
|
||||
|
||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
||||
// read the sleep types from it.
|
||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||
for (0..aml_block_count) |i| {
|
||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||
}
|
||||
const active = blocks[0..aml_block_count];
|
||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||
namespace = pr.namespace;
|
||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||
// The namespace's Device objects are no longer folded into the kernel
|
||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||
// (published below), re-parses the same blobs, and registers + reports
|
||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||
// \_S5 sleep type above. The device-building helpers below
|
||||
// (wireAcpiDevices and friends) are retained but unreferenced — a
|
||||
// focused dead-code sweep follows the migration.
|
||||
} else |_| {
|
||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||
// whatever the FADT alone provided.
|
||||
}
|
||||
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
|
||||
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
|
||||
// and the power register map. The AML bytecode (device enumeration and the
|
||||
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
|
||||
// job: it claims the acpi-tables node published below, parses the same blobs,
|
||||
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
|
||||
// AML in the kernel keeps boot latency off the critical, single-core path.
|
||||
|
||||
// Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as
|
||||
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||
// claimant. Kept even when the kernel-side device building (above) retires
|
||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
|
||||
// now that the kernel keeps only the *static* tables for itself.
|
||||
publishAcpiTablesNode(device_tree) catch {};
|
||||
}
|
||||
|
||||
@@ -442,7 +420,7 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||
// that names them is parsed, so the grant is the whole space — the honest
|
||||
// trust boundary of docs/m19-m20-plan.md decision 5.
|
||||
// trust boundary of docs/discovery.md (the acpi service's one trusted node).
|
||||
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||
@@ -450,12 +428,9 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
|
||||
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
|
||||
_ = node.addResource(.irq, 0, 256);
|
||||
}
|
||||
|
||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||
pub fn amlDeviceCount() usize {
|
||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||
return 0;
|
||||
// The FADT rides along (M21): the service reads the PM1 event / GPE blocks
|
||||
// from its own copy, telling it apart from the AML blobs by signature.
|
||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||
}
|
||||
|
||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||
@@ -485,13 +460,15 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
||||
try parseHpet(device_tree, hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
||||
fadt_physical = sdt_physical;
|
||||
fadt_length = header.length;
|
||||
parseFadt(header);
|
||||
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||
parseSpcr(header);
|
||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||
parseDmar(hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||
addAmlBlock(sdt_physical);
|
||||
}
|
||||
// Any other signature is recognised but left opaque for now.
|
||||
@@ -599,7 +576,7 @@ fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeade
|
||||
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||
|
||||
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||
/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR
|
||||
/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR
|
||||
/// resources, and `device_register` containment demands the bridge own windows
|
||||
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||
@@ -729,8 +706,8 @@ const fadt_x_pm_tmr_blk = 208; // GAS
|
||||
const flag_reset_register_supported = 1 << 10;
|
||||
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||
|
||||
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
||||
/// FADT -> the power register map (into `power_information`) and the DSDT address,
|
||||
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
|
||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const len: usize = header.length;
|
||||
@@ -816,342 +793,6 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
}
|
||||
}
|
||||
|
||||
// --- AML namespace -> generic device tree -----------------------------------
|
||||
|
||||
/// The PCI bus context while descending the ACPI namespace: the generic host
|
||||
/// bridge whose children ACPI address (`_ADR`) devices resolve against, and the bus number.
|
||||
const PciContext = struct { bridge: *device_model.Device, bus: u8 };
|
||||
|
||||
/// Mirror the ACPI namespace's Device objects into the generic tree, *merging*
|
||||
/// them with the PCI-enumerated nodes: a PCI root bridge (`PNP0A03`/`PNP0A08`)
|
||||
/// folds onto the existing `pci_host_bridge`, and each addressed (`_ADR`) device folds onto
|
||||
/// the matching PCI function (annotating it with the ACPI hardware ID (`_HID`) and nesting the
|
||||
/// ACPI-only children — keyboard, RTC, … — beneath it). Namespace devices with no
|
||||
/// PCI match land under a synthetic `acpi` node.
|
||||
fn wireAcpiDevices(device_tree: *DeviceTree, aml_namespace: *aml.Namespace, hal: Hal) !void {
|
||||
var arena = std.heap.ArenaAllocator.init(device_tree.allocator);
|
||||
defer arena.deinit();
|
||||
var interpreter = aml.Interpreter.init(aml_namespace, .{
|
||||
.mapMmio = hal.mapMmio,
|
||||
.pioRead = hal.pioRead,
|
||||
.pioWrite = hal.pioWrite,
|
||||
}, arena.allocator());
|
||||
|
||||
const acpi_root = try device_tree.addChild(device_tree.root, .unknown, "acpi");
|
||||
try mirrorDevices(device_tree, aml_namespace.root, acpi_root, null, &interpreter);
|
||||
}
|
||||
|
||||
fn mirrorDevices(device_tree: *DeviceTree, node: *aml.Node, parent_device: *device_model.Device, context: ?PciContext, interpreter: *aml.Interpreter) (error{OutOfMemory})!void {
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) {
|
||||
if (c.kind != .device) {
|
||||
// A scope — the System Bus (\_SB), General Purpose Events (\_GPE), … —
|
||||
// descend without adding a node.
|
||||
try mirrorDevices(device_tree, c, parent_device, context, interpreter);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip devices the firmware reports as not present (via a device-status (`_STA`) method),
|
||||
// along with their whole subtree — per the ACPI rules.
|
||||
if (!devicePresent(interpreter, c)) continue;
|
||||
|
||||
var mirrored_device: *device_model.Device = undefined;
|
||||
var child_context = context;
|
||||
|
||||
if (isPciRootNode(c)) {
|
||||
// The PCI root bridge folds onto the generic host bridge.
|
||||
mirrored_device = matchHostBridge(device_tree) orelse
|
||||
try device_tree.addChild(parent_device, .acpi_device, &c.segment);
|
||||
child_context = .{ .bridge = mirrored_device, .bus = 0 };
|
||||
} else {
|
||||
// An addressed device folds onto its matching PCI function; anything
|
||||
// else becomes a fresh node under the current parent.
|
||||
mirrored_device = pick: {
|
||||
if (context) |pc| {
|
||||
if (readAdr(c)) |adr| {
|
||||
if (findPciNode(pc.bridge, pc.bus, adr)) |pnode| break :pick pnode;
|
||||
}
|
||||
}
|
||||
break :pick try device_tree.addChild(parent_device, .acpi_device, &c.segment);
|
||||
};
|
||||
}
|
||||
|
||||
applyHid(mirrored_device, c, interpreter);
|
||||
applyCrs(mirrored_device, c, interpreter);
|
||||
try mirrorDevices(device_tree, c, mirrored_device, child_context, interpreter);
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate a device's status (`_STA`) to decide if it is present. An absent status
|
||||
/// (`_STA`) means present by default; an evaluation failure is treated as present too (we'd
|
||||
/// rather over-report than hide a device we couldn't introspect).
|
||||
fn devicePresent(interpreter: *aml.Interpreter, node: *aml.Node) bool {
|
||||
const sta = aml.Namespace.childOf(node, seg4("_STA")) orelse return true;
|
||||
const obj = interpreter.evaluate(sta, &.{}) catch return true;
|
||||
const status = obj.asInteger() catch return true;
|
||||
return (status & 0x01) != 0; // bit 0 = present
|
||||
}
|
||||
|
||||
/// The first PCI host bridge in the generic tree (segment 0).
|
||||
fn matchHostBridge(device_tree: *DeviceTree) ?*device_model.Device {
|
||||
var c = device_tree.root.first_child;
|
||||
while (c) |ch| : (c = ch.next_sibling) {
|
||||
if (ch.class == .pci_host_bridge) return ch;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The PCI function node under `bridge` at the address the device's address object
|
||||
/// (`_ADR`) names (device/function on
|
||||
/// `bus`), or null.
|
||||
fn findPciNode(bridge: *device_model.Device, bus: u8, adr: u32) ?*device_model.Device {
|
||||
const device: u16 = @truncate((adr >> 16) & 0x1F);
|
||||
const function: u16 = @truncate(adr & 0x7);
|
||||
const target: u16 = (@as(u16, bus) << 8) | (device << 3) | function;
|
||||
var c = bridge.first_child;
|
||||
while (c) |ch| : (c = ch.next_sibling) {
|
||||
if (ch.ids.pci_bdf) |bdf| {
|
||||
if (bdf == target) return ch;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// A device's address (`_ADR`) — a static integer Name — or null.
|
||||
fn readAdr(node: *aml.Node) ?u32 {
|
||||
const n = aml.Namespace.childOf(node, seg4("_ADR")) orelse return null;
|
||||
if (n.kind != .name) return null;
|
||||
var p: usize = 0;
|
||||
return @truncate(readIntObj(n.value, &p) orelse return null);
|
||||
}
|
||||
|
||||
/// Whether a `_HID` string names a PCI(e) host bridge.
|
||||
fn isPciRootHid(hid: []const u8) bool {
|
||||
const id = acpi_ids.HardwareId.fromHid(hid) orelse return false;
|
||||
return id == .pci_bus or id == .pci_express_root_bridge;
|
||||
}
|
||||
|
||||
/// Whether a namespace device is a PCI(e) host bridge. A packed EISA id is decoded
|
||||
/// to its string form first, so both encodings answer through the one registry.
|
||||
fn isPciRootNode(node: *aml.Node) bool {
|
||||
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return false;
|
||||
if (hid.kind != .name or hid.value.len == 0) return false;
|
||||
switch (hid.value[0]) {
|
||||
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||
var p: usize = 0;
|
||||
const n = readIntObj(hid.value, &p) orelse return false;
|
||||
var buffer: [8]u8 = undefined;
|
||||
return isPciRootHid(eisaIdToStr(@truncate(n), &buffer));
|
||||
},
|
||||
0x0D => return isPciRootHid(cstr(hid.value[1..])),
|
||||
else => return false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a device's hardware ID (`_HID`) into the generic device: an integer decodes as an EISA
|
||||
/// id ("PNP0A03"), a string is taken verbatim. Handles both the common static
|
||||
/// Name form and a Method form (evaluated).
|
||||
fn applyHid(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
const hid = aml.Namespace.childOf(node, seg4("_HID")) orelse return;
|
||||
if (hid.kind == .method) {
|
||||
const obj = interpreter.evaluate(hid, &.{}) catch return;
|
||||
switch (obj) {
|
||||
.integer => |n| setEisaHid(device, @truncate(n)),
|
||||
.string => |s| device.setHid(s),
|
||||
else => {},
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (hid.kind != .name or hid.value.len == 0) return;
|
||||
const v = hid.value;
|
||||
switch (v[0]) {
|
||||
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||
var p: usize = 0;
|
||||
const n = readIntObj(v, &p) orelse return;
|
||||
setEisaHid(device, @truncate(n));
|
||||
},
|
||||
0x0D => device.setHid(cstr(v[1..])), // StringPrefix
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
fn setEisaHid(device: *device_model.Device, id: u32) void {
|
||||
device.ids.acpi_hid = id;
|
||||
var buffer: [8]u8 = undefined;
|
||||
device.setHid(eisaIdToStr(id, &buffer));
|
||||
}
|
||||
|
||||
/// Parse a device's current resource settings (`_CRS`). The evaluator handles both the static
|
||||
/// `Buffer` form (a `Name`) and the method form uniformly, yielding the
|
||||
/// ResourceTemplate bytes we then decode.
|
||||
fn applyCrs(device: *device_model.Device, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
||||
const obj = interpreter.evaluate(crs, &.{}) catch return;
|
||||
const buffer = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
parseResourceTemplate(device, buffer);
|
||||
}
|
||||
|
||||
/// Walk a ResourceTemplate byte list, adding recognised descriptors as resources.
|
||||
fn parseResourceTemplate(device: *device_model.Device, bytes: []const u8) void {
|
||||
var i: usize = 0;
|
||||
while (i < bytes.len) {
|
||||
const tag = bytes[i];
|
||||
if (tag & 0x80 == 0) {
|
||||
// Small descriptor: length in low 3 bits, type in bits [6:3].
|
||||
const len: usize = tag & 0x07;
|
||||
const body = i + 1;
|
||||
if (body + len > bytes.len) break;
|
||||
switch ((tag >> 3) & 0x0F) {
|
||||
0x04 => if (len >= 2) { // IRQ: a 16-bit mask, one resource per set bit
|
||||
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
||||
var b: usize = 0;
|
||||
while (b < 16) : (b += 1) {
|
||||
if (mask & (@as(u16, 1) << @intCast(b)) != 0) _ = device.addResource(.irq, b, 1);
|
||||
}
|
||||
},
|
||||
0x08 => if (len >= 7) { // IO port: minimum at +1, length at +6
|
||||
_ = device.addResource(.io_port, rd16(bytes, body + 1), bytes[body + 6]);
|
||||
},
|
||||
0x09 => if (len >= 3) { // Fixed IO: base at +0, length at +2
|
||||
_ = device.addResource(.io_port, rd16(bytes, body), bytes[body + 2]);
|
||||
},
|
||||
0x0F => break, // EndTag
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
} else {
|
||||
// Large descriptor: 16-bit length follows the tag.
|
||||
if (i + 3 > bytes.len) break;
|
||||
const len: usize = @intCast(rd16(bytes, i + 1));
|
||||
const body = i + 3;
|
||||
if (body + len > bytes.len) break;
|
||||
switch (tag) {
|
||||
0x85 => if (len >= 17) { // Memory32: minimum at +1, length at +13
|
||||
_ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 13));
|
||||
},
|
||||
0x86 => if (len >= 9) { // Memory32Fixed: base at +1, length at +5
|
||||
_ = device.addResource(.memory, rd32(bytes, body + 1), rd32(bytes, body + 5));
|
||||
},
|
||||
0x89 => if (len >= 2) { // Extended IRQ: count at +1, then count u32s
|
||||
const count = bytes[body + 1];
|
||||
var k: usize = 0;
|
||||
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
||||
_ = device.addResource(.irq, rd32(bytes, body + 2 + k * 4), 1);
|
||||
}
|
||||
},
|
||||
0x87, 0x88, 0x8A => parseAddressSpace(device, tag, bytes[body .. body + len]),
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Word/DWord/QWord address-space descriptors: resource type at [0], then
|
||||
/// granularity/minimum/maximum/translation/length, each of width `w`.
|
||||
fn parseAddressSpace(device: *device_model.Device, tag: u8, body: []const u8) void {
|
||||
const w: usize = switch (tag) {
|
||||
0x88 => 2, // Word
|
||||
0x87 => 4, // DWord
|
||||
else => 8, // QWord (0x8A)
|
||||
};
|
||||
if (body.len < 3 + 5 * w) return;
|
||||
const minimum = readN(body, 3 + w, w);
|
||||
const length = readN(body, 3 + 4 * w, w);
|
||||
const kind: device_model.ResourceKind = switch (body[0]) {
|
||||
0 => .memory,
|
||||
1 => .io_port,
|
||||
else => .bus_range,
|
||||
};
|
||||
_ = device.addResource(kind, minimum, length);
|
||||
}
|
||||
|
||||
/// Decode a packed EISA id into its 7-char string (e.g. 0x030AD041 -> "PNP0A03").
|
||||
fn eisaIdToStr(id: u32, buffer: *[8]u8) []const u8 {
|
||||
const b0: u16 = @intCast(id & 0xFF);
|
||||
const b1: u16 = @intCast((id >> 8) & 0xFF);
|
||||
const b2: u8 = @truncate(id >> 16);
|
||||
const b3: u8 = @truncate(id >> 24);
|
||||
const mfg = (b0 << 8) | b1;
|
||||
buffer[0] = '@' + @as(u8, @intCast((mfg >> 10) & 0x1F));
|
||||
buffer[1] = '@' + @as(u8, @intCast((mfg >> 5) & 0x1F));
|
||||
buffer[2] = '@' + @as(u8, @intCast(mfg & 0x1F));
|
||||
buffer[3] = hexDigit((b2 >> 4) & 0xF);
|
||||
buffer[4] = hexDigit(b2 & 0xF);
|
||||
buffer[5] = hexDigit((b3 >> 4) & 0xF);
|
||||
buffer[6] = hexDigit(b3 & 0xF);
|
||||
return buffer[0..7];
|
||||
}
|
||||
|
||||
fn hexDigit(n: u8) u8 {
|
||||
return if (n < 10) '0' + n else 'A' + (n - 10);
|
||||
}
|
||||
|
||||
fn seg4(comptime s: *const [4:0]u8) [4]u8 {
|
||||
return s[0..4].*;
|
||||
}
|
||||
|
||||
fn cstr(bytes: []const u8) []const u8 {
|
||||
const index = std.mem.indexOfScalar(u8, bytes, 0) orelse bytes.len;
|
||||
return bytes[0..index];
|
||||
}
|
||||
|
||||
const PkgLen = struct { value: usize, size: usize };
|
||||
|
||||
fn packageLength(bytes: []const u8, p: usize) ?PkgLen {
|
||||
if (p >= bytes.len) return null;
|
||||
const lead = bytes[p];
|
||||
const follow: usize = lead >> 6;
|
||||
if (p + 1 + follow > bytes.len) return null;
|
||||
if (follow == 0) return .{ .value = lead & 0x3F, .size = 1 };
|
||||
var value: usize = lead & 0x0F;
|
||||
var i: usize = 0;
|
||||
while (i < follow) : (i += 1) value |= @as(usize, bytes[p + 1 + i]) << @intCast(4 + i * 8);
|
||||
return .{ .value = value, .size = 1 + follow };
|
||||
}
|
||||
|
||||
/// Read an AML integer object at `p`, advancing `p` past it.
|
||||
fn readIntObj(bytes: []const u8, p: *usize) ?u64 {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const opcode = bytes[p.*];
|
||||
p.* += 1;
|
||||
return switch (opcode) {
|
||||
0x00 => 0,
|
||||
0x01 => 1,
|
||||
0xFF => 0xFF,
|
||||
0x0A => readLE(bytes, p, 1),
|
||||
0x0B => readLE(bytes, p, 2),
|
||||
0x0C => readLE(bytes, p, 4),
|
||||
0x0E => readLE(bytes, p, 8),
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
fn readLE(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
||||
if (p.* + n > bytes.len) return null;
|
||||
const v = readN(bytes, p.*, n);
|
||||
p.* += n;
|
||||
return v;
|
||||
}
|
||||
|
||||
fn readN(bytes: []const u8, off: usize, n: usize) u64 {
|
||||
var v: u64 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < n and off + k < bytes.len) : (k += 1) v |= @as(u64, bytes[off + k]) << @intCast(k * 8);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn rd16(bytes: []const u8, off: usize) u64 {
|
||||
return readN(bytes, off, 2);
|
||||
}
|
||||
|
||||
fn rd32(bytes: []const u8, off: usize) u64 {
|
||||
return readN(bytes, off, 4);
|
||||
}
|
||||
|
||||
// --- helpers ----------------------------------------------------------------
|
||||
|
||||
/// Sum `len` bytes; an ACPI table/pointer is valid when the low 8 bits are zero.
|
||||
@@ -1192,42 +833,9 @@ fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_o
|
||||
return .{ .mmio = false, .address = port, .width = width };
|
||||
}
|
||||
|
||||
/// The mapped configuration space of one PCI function (its 4 KiB ECAM page). Mapped
|
||||
/// writable so BAR sizing can probe it; reads and writes both go through here.
|
||||
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||
/// x86 is little-endian and native, so an unaligned load suffices.
|
||||
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
||||
const p: *align(1) const T = @ptrCast(bytes + off);
|
||||
return p.*;
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
test "eisaIdToStr decodes a packed EISA id" {
|
||||
var buffer: [8]u8 = undefined;
|
||||
// 0x030AD041 is the well-known encoding of "PNP0A03" (PCI root bridge).
|
||||
try std.testing.expectEqualStrings("PNP0A03", eisaIdToStr(0x030AD041, &buffer));
|
||||
}
|
||||
|
||||
test "parseResourceTemplate extracts IO, IRQ, and fixed memory" {
|
||||
// ResourceTemplate { IO(minimum 0x60, len 8), IRQ(4), Memory32Fixed(0xFED00000, 0x1000) }
|
||||
const runtime = [_]u8{
|
||||
0x47, 0x01, 0x60, 0x00, 0x60, 0x00, 0x01, 0x08, // small IO descriptor
|
||||
0x22, 0x10, 0x00, // small IRQ descriptor (mask bit 4 -> IRQ 4)
|
||||
0x86, 0x09, 0x00, 0x01, 0x00, 0x00, 0xD0, 0xFE, 0x00, 0x10, 0x00, 0x00, // Memory32Fixed
|
||||
0x79, 0x00, // EndTag
|
||||
};
|
||||
var device = device_model.Device{};
|
||||
parseResourceTemplate(&device, &runtime);
|
||||
|
||||
try std.testing.expectEqual(@as(u8, 3), device.resource_count);
|
||||
const rs = device.resources[0..device.resource_count];
|
||||
try std.testing.expectEqual(device_model.ResourceKind.io_port, rs[0].kind);
|
||||
try std.testing.expectEqual(@as(u64, 0x60), rs[0].start);
|
||||
try std.testing.expectEqual(@as(u64, 8), rs[0].len);
|
||||
try std.testing.expectEqual(device_model.ResourceKind.irq, rs[1].kind);
|
||||
try std.testing.expectEqual(@as(u64, 4), rs[1].start);
|
||||
try std.testing.expectEqual(device_model.ResourceKind.memory, rs[2].kind);
|
||||
try std.testing.expectEqual(@as(u64, 0xFED00000), rs[2].start);
|
||||
try std.testing.expectEqual(@as(u64, 0x1000), rs[2].len);
|
||||
}
|
||||
|
||||
@@ -12,6 +12,11 @@ const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const parser = @import("parser.zig");
|
||||
|
||||
/// The named AML opcode/prefix bytes (`zero_opcode`, `byte_prefix`, …). Re-exported so
|
||||
/// callers that decode raw AML bytes — e.g. the acpi service reading a `_HID` integer —
|
||||
/// name the opcodes instead of writing bare 0x0A/0x0B/… literals (docs/coding-standards.md).
|
||||
pub const opcodes = @import("opcodes.zig");
|
||||
|
||||
pub const Namespace = @import("namespace.zig").Namespace;
|
||||
pub const Node = @import("namespace.zig").Node;
|
||||
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||
@@ -51,7 +56,7 @@ pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseRes
|
||||
}
|
||||
|
||||
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||
/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts
|
||||
/// (docs/discovery.md) reports, and what the kernel's own parse counts
|
||||
/// so the two can be checked equal across the ring-3 move.
|
||||
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||
return countKind(namespace.root, .device);
|
||||
@@ -225,3 +230,31 @@ test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||
}
|
||||
|
||||
test "interpreter records Notify(device, code)" {
|
||||
// Device(DEV_) { Name(_HID, 0x030AD041) } // PNP0A03-ish placeholder
|
||||
// Method(TST_, 0) { Notify(DEV_, 0x80); Return(Zero) }
|
||||
// Encoded: a Device holding a Name, then a Method issuing Notify on it.
|
||||
const blob = [_]u8{
|
||||
0x5B, 0x82, 0x0F, 0x44, 0x45, 0x56, 0x5F, // Device(DEV_) len=0x0F (pkglen + DEV_ + Name)
|
||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0C, 0x41, 0xD0, 0x0A, 0x03, // Name(_HID, DWord 0x030AD041)
|
||||
0x14, 0x0F, 0x54, 0x53, 0x54, 0x5F, 0x00, // Method(TST_, 0) len=0x0F (pkglen + TST_ + flags + body)
|
||||
0x86, 0x44, 0x45, 0x56, 0x5F, 0x0A, 0x80, // Notify(DEV_, 0x80)
|
||||
0xA4, 0x00, // Return(Zero)
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
const namespace = &result.namespace;
|
||||
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||
const dev = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'E', 'V', '_' }}) orelse return error.NoDevice;
|
||||
|
||||
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||
_ = try interpreter.evaluate(tst, &.{});
|
||||
|
||||
const events = interpreter.takeNotifications();
|
||||
try std.testing.expectEqual(@as(usize, 1), events.len);
|
||||
try std.testing.expectEqual(dev, events[0].node);
|
||||
try std.testing.expectEqual(@as(u64, 0x80), events[0].code);
|
||||
}
|
||||
|
||||
@@ -141,6 +141,9 @@ const Frame = struct {
|
||||
/// A CreateField binding: a name that indexes into a buffer object.
|
||||
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||
|
||||
/// One Notify(device, code) the interpreter executed.
|
||||
pub const NotifyEvent = struct { node: *Node, code: u64 };
|
||||
|
||||
pub const Interpreter = struct {
|
||||
namespace: *Namespace,
|
||||
hal: Hal,
|
||||
@@ -149,6 +152,11 @@ pub const Interpreter = struct {
|
||||
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||
/// CreateField bindings active for the current evaluation.
|
||||
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||
/// Notify(device, code) operations the last evaluation executed — a GPE or
|
||||
/// EC handler tells the OS "look at this device" this way. Bounded; the
|
||||
/// caller drains it with `takeNotifications` after `evaluate` (M21).
|
||||
notify_queue: [16]NotifyEvent = undefined,
|
||||
notify_count: usize = 0,
|
||||
|
||||
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||
@@ -157,6 +165,7 @@ pub const Interpreter = struct {
|
||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||
/// Field. Resets per-evaluation runtime state first.
|
||||
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||
self.notify_count = 0;
|
||||
self.dynamic_overrides.clearRetainingCapacity();
|
||||
self.fields.clearRetainingCapacity();
|
||||
return self.invoke(node, args);
|
||||
@@ -267,6 +276,8 @@ pub const Interpreter = struct {
|
||||
},
|
||||
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||
|
||||
opcode.notify_opcode => try self.notify(current, frame),
|
||||
|
||||
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||
|
||||
// CreateXField: source, index, name (bit widths differ by op)
|
||||
@@ -542,6 +553,36 @@ pub const Interpreter = struct {
|
||||
try self.storeInto(current, frame, value);
|
||||
}
|
||||
|
||||
/// Notify(SuperName, NotifyValue): resolve the named device, evaluate the
|
||||
/// code, and record the pair for the caller to dispatch. AML control flow
|
||||
/// continues (Notify returns nothing).
|
||||
fn notify(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
var target: ?*Node = null;
|
||||
if (isNameStart(lead)) {
|
||||
const name_path = try current.nameString();
|
||||
target = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||
} else {
|
||||
// A non-name SuperName (Local/Arg holding a reference).
|
||||
const obj = try self.term(current, frame);
|
||||
if (obj == .reference) target = obj.reference;
|
||||
}
|
||||
const code = try self.evaluateInteger(current, frame);
|
||||
if (target) |node| {
|
||||
if (self.notify_count < self.notify_queue.len) {
|
||||
self.notify_queue[self.notify_count] = .{ .node = node, .code = code };
|
||||
self.notify_count += 1;
|
||||
}
|
||||
}
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
/// The Notify events the last `evaluate` produced. Valid until the next
|
||||
/// `evaluate` clears the queue.
|
||||
pub fn takeNotifications(self: *Interpreter) []const NotifyEvent {
|
||||
return self.notify_queue[0..self.notify_count];
|
||||
}
|
||||
|
||||
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) {
|
||||
|
||||
@@ -29,10 +29,15 @@ pub const DeviceClass = enum(u32) {
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||
/// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs,
|
||||
/// service (docs/discovery.md): memory resources over the AML blobs,
|
||||
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||
acpi_tables,
|
||||
/// One interface of a USB device, registered by the xHCI bus driver. It owns
|
||||
/// no MMIO — it is reached through its controller — so it carries no
|
||||
/// resources; the (class, subclass, protocol) triple that says what it is
|
||||
/// travels in the bus report's identity, not here.
|
||||
usb_device,
|
||||
unknown,
|
||||
};
|
||||
|
||||
|
||||
+496
-189
@@ -7,6 +7,16 @@
|
||||
//! apart. Pure reference data (from the PCI spec; see https://wiki.osdev.org/PCI) — no
|
||||
//! hardware access — so it is shared by kernel discovery (the device-tree dump) and any
|
||||
//! user-space tool (a future lspci, driver matching).
|
||||
//!
|
||||
//! The taxonomy is named, not numbered (docs/coding-standards.md, "Named values"): the
|
||||
//! base class is a `BaseClass` enum, and each class with defined subclasses gets a
|
||||
//! namespace holding its `SubClass` enum (and, where the spec defines them, per-subclass
|
||||
//! `ProgIf` enums) — the same shape as `usb-ids.zig`. Code that *means* a specific class
|
||||
//! names it (`BaseClass.serial_bus`, `serial_bus.usb.ProgIf.xhci`) rather than writing a
|
||||
//! bare 0x0C/0x03/0x30. The `className`/`subclassName`/`progIfName` functions still take
|
||||
//! the raw bytes a function reports in its header, because that is what hardware hands us.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// The three bytes of a PCI class code, unpacked from the `0xCCSSPP` value discovery
|
||||
/// records in `Device.ids.pci_class` (CC = base class, SS = subclass, PP = prog-IF).
|
||||
@@ -22,148 +32,465 @@ pub const ClassCode = struct {
|
||||
.prog_if = @intCast(packed_code & 0xFF),
|
||||
};
|
||||
}
|
||||
|
||||
/// Re-pack the triple into the `0xCCSSPP` form. Lets code name a whole class code
|
||||
/// from its parts — `pack(.{ .base = @intFromEnum(BaseClass.serial_bus), … })` —
|
||||
/// instead of writing the literal 0x0C0330.
|
||||
pub fn pack(self: ClassCode) u24 {
|
||||
return (@as(u24, self.base) << 16) | (@as(u24, self.subclass) << 8) | self.prog_if;
|
||||
}
|
||||
};
|
||||
|
||||
/// Base class (config byte 0x0B). Non-exhaustive: an unlisted code is a real but
|
||||
/// unnamed class, decoded as "Unknown" rather than rejected.
|
||||
pub const BaseClass = enum(u8) {
|
||||
unclassified = 0x00,
|
||||
mass_storage = 0x01,
|
||||
network = 0x02,
|
||||
display = 0x03,
|
||||
multimedia = 0x04,
|
||||
memory = 0x05,
|
||||
bridge = 0x06,
|
||||
simple_communication = 0x07,
|
||||
base_system_peripheral = 0x08,
|
||||
input_device = 0x09,
|
||||
docking_station = 0x0A,
|
||||
processor = 0x0B,
|
||||
serial_bus = 0x0C,
|
||||
wireless = 0x0D,
|
||||
intelligent = 0x0E,
|
||||
satellite_communication = 0x0F,
|
||||
encryption = 0x10,
|
||||
signal_processing = 0x11,
|
||||
processing_accelerator = 0x12,
|
||||
non_essential_instrumentation = 0x13,
|
||||
co_processor = 0x40,
|
||||
unassigned = 0xFF,
|
||||
_,
|
||||
|
||||
pub fn name(self: BaseClass) []const u8 {
|
||||
return switch (self) {
|
||||
.unclassified => "Unclassified",
|
||||
.mass_storage => "Mass Storage Controller",
|
||||
.network => "Network Controller",
|
||||
.display => "Display Controller",
|
||||
.multimedia => "Multimedia Controller",
|
||||
.memory => "Memory Controller",
|
||||
.bridge => "Bridge",
|
||||
.simple_communication => "Simple Communication Controller",
|
||||
.base_system_peripheral => "Base System Peripheral",
|
||||
.input_device => "Input Device Controller",
|
||||
.docking_station => "Docking Station",
|
||||
.processor => "Processor",
|
||||
.serial_bus => "Serial Bus Controller",
|
||||
.wireless => "Wireless Controller",
|
||||
.intelligent => "Intelligent Controller",
|
||||
.satellite_communication => "Satellite Communication Controller",
|
||||
.encryption => "Encryption Controller",
|
||||
.signal_processing => "Signal Processing Controller",
|
||||
.processing_accelerator => "Processing Accelerator",
|
||||
.non_essential_instrumentation => "Non-Essential Instrumentation",
|
||||
.co_processor => "Co-Processor",
|
||||
.unassigned => "Unassigned Class (Vendor specific)",
|
||||
_ => "Unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// --- Per-class subclass (and prog-IF) taxonomies --------------------------------------
|
||||
// One namespace per base class that has defined subclasses, named after the class. Each
|
||||
// holds an exhaustive `SubClass` enum (so an unlisted code decodes to the class default,
|
||||
// not a wrong name), and, where the spec assigns them, per-subclass `ProgIf` enums.
|
||||
|
||||
pub const mass_storage = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
scsi_bus = 0x00,
|
||||
ide = 0x01,
|
||||
floppy = 0x02,
|
||||
ipi_bus = 0x03,
|
||||
raid = 0x04,
|
||||
ata = 0x05,
|
||||
serial_ata = 0x06,
|
||||
serial_attached_scsi = 0x07,
|
||||
non_volatile_memory = 0x08,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.scsi_bus => "SCSI Bus Controller",
|
||||
.ide => "IDE Controller",
|
||||
.floppy => "Floppy Disk Controller",
|
||||
.ipi_bus => "IPI Bus Controller",
|
||||
.raid => "RAID Controller",
|
||||
.ata => "ATA Controller",
|
||||
.serial_ata => "Serial ATA Controller",
|
||||
.serial_attached_scsi => "Serial Attached SCSI Controller",
|
||||
.non_volatile_memory => "Non-Volatile Memory Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const serial_ata = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
vendor_specific = 0x00,
|
||||
ahci = 0x01,
|
||||
serial_storage_bus = 0x02,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.vendor_specific => "Vendor Specific Interface",
|
||||
.ahci => "AHCI 1.0",
|
||||
.serial_storage_bus => "Serial Storage Bus",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const non_volatile_memory = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
nvmhci = 0x01,
|
||||
nvm_express = 0x02,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.nvmhci => "NVMHCI",
|
||||
.nvm_express => "NVM Express",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const network = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
ethernet = 0x00,
|
||||
token_ring = 0x01,
|
||||
fddi = 0x02,
|
||||
atm = 0x03,
|
||||
isdn = 0x04,
|
||||
picmg_multi_computing = 0x06,
|
||||
infiniband = 0x07,
|
||||
fabric = 0x08,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.ethernet => "Ethernet Controller",
|
||||
.token_ring => "Token Ring Controller",
|
||||
.fddi => "FDDI Controller",
|
||||
.atm => "ATM Controller",
|
||||
.isdn => "ISDN Controller",
|
||||
.picmg_multi_computing => "PICMG 2.14 Multi Computing Controller",
|
||||
.infiniband => "Infiniband Controller",
|
||||
.fabric => "Fabric Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const display = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
vga_compatible = 0x00,
|
||||
xga = 0x01,
|
||||
three_dimensional = 0x02,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.vga_compatible => "VGA Compatible Controller",
|
||||
.xga => "XGA Controller",
|
||||
.three_dimensional => "3D Controller (Not VGA-Compatible)",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const vga_compatible = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
vga = 0x00,
|
||||
compatible_8514 = 0x01,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.vga => "VGA Controller",
|
||||
.compatible_8514 => "8514-Compatible Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const multimedia = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
video = 0x00,
|
||||
audio = 0x01,
|
||||
telephony = 0x02,
|
||||
audio_device = 0x03,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.video => "Multimedia Video Controller",
|
||||
.audio => "Multimedia Audio Controller",
|
||||
.telephony => "Computer Telephony Device",
|
||||
.audio_device => "Audio Device",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const memory = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
ram = 0x00,
|
||||
flash = 0x01,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.ram => "RAM Controller",
|
||||
.flash => "Flash Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const bridge = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
host = 0x00,
|
||||
isa = 0x01,
|
||||
eisa = 0x02,
|
||||
mca = 0x03,
|
||||
pci_to_pci = 0x04,
|
||||
pcmcia = 0x05,
|
||||
nubus = 0x06,
|
||||
cardbus = 0x07,
|
||||
raceway = 0x08,
|
||||
pci_to_pci_semi_transparent = 0x09,
|
||||
infiniband_to_pci = 0x0A,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.host => "Host Bridge",
|
||||
.isa => "ISA Bridge",
|
||||
.eisa => "EISA Bridge",
|
||||
.mca => "MCA Bridge",
|
||||
.pci_to_pci => "PCI-to-PCI Bridge",
|
||||
.pcmcia => "PCMCIA Bridge",
|
||||
.nubus => "NuBus Bridge",
|
||||
.cardbus => "CardBus Bridge",
|
||||
.raceway => "RACEway Bridge",
|
||||
.pci_to_pci_semi_transparent => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||
.infiniband_to_pci => "InfiniBand-to-PCI Host Bridge",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const pci_to_pci = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
normal_decode = 0x00,
|
||||
subtractive_decode = 0x01,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.normal_decode => "Normal Decode",
|
||||
.subtractive_decode => "Subtractive Decode",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const simple_communication = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
serial = 0x00,
|
||||
parallel = 0x01,
|
||||
multiport_serial = 0x02,
|
||||
modem = 0x03,
|
||||
gpib = 0x04,
|
||||
smart_card = 0x05,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.serial => "Serial Controller",
|
||||
.parallel => "Parallel Controller",
|
||||
.multiport_serial => "Multiport Serial Controller",
|
||||
.modem => "Modem",
|
||||
.gpib => "IEEE 488.1/2 (GPIB) Controller",
|
||||
.smart_card => "Smart Card Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const serial = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
compatible_8250 = 0x00,
|
||||
compatible_16450 = 0x01,
|
||||
compatible_16550 = 0x02,
|
||||
compatible_16650 = 0x03,
|
||||
compatible_16750 = 0x04,
|
||||
compatible_16850 = 0x05,
|
||||
compatible_16950 = 0x06,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.compatible_8250 => "8250-Compatible (Generic XT)",
|
||||
.compatible_16450 => "16450-Compatible",
|
||||
.compatible_16550 => "16550-Compatible",
|
||||
.compatible_16650 => "16650-Compatible",
|
||||
.compatible_16750 => "16750-Compatible",
|
||||
.compatible_16850 => "16850-Compatible",
|
||||
.compatible_16950 => "16950-Compatible",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const base_system_peripheral = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
pic = 0x00,
|
||||
dma = 0x01,
|
||||
timer = 0x02,
|
||||
rtc = 0x03,
|
||||
pci_hot_plug = 0x04,
|
||||
sd_host = 0x05,
|
||||
iommu = 0x06,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.pic => "PIC",
|
||||
.dma => "DMA Controller",
|
||||
.timer => "Timer",
|
||||
.rtc => "RTC Controller",
|
||||
.pci_hot_plug => "PCI Hot-Plug Controller",
|
||||
.sd_host => "SD Host Controller",
|
||||
.iommu => "IOMMU",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const input_device = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
keyboard = 0x00,
|
||||
digitizer_pen = 0x01,
|
||||
mouse = 0x02,
|
||||
scanner = 0x03,
|
||||
gameport = 0x04,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.keyboard => "Keyboard Controller",
|
||||
.digitizer_pen => "Digitizer Pen",
|
||||
.mouse => "Mouse Controller",
|
||||
.scanner => "Scanner Controller",
|
||||
.gameport => "Gameport Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
pub const serial_bus = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
firewire = 0x00,
|
||||
access_bus = 0x01,
|
||||
ssa = 0x02,
|
||||
usb = 0x03,
|
||||
fibre_channel = 0x04,
|
||||
smbus = 0x05,
|
||||
infiniband = 0x06,
|
||||
ipmi = 0x07,
|
||||
sercos = 0x08,
|
||||
canbus = 0x09,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.firewire => "FireWire (IEEE 1394) Controller",
|
||||
.access_bus => "ACCESS Bus Controller",
|
||||
.ssa => "SSA",
|
||||
.usb => "USB Controller",
|
||||
.fibre_channel => "Fibre Channel",
|
||||
.smbus => "SMBus Controller",
|
||||
.infiniband => "InfiniBand Controller",
|
||||
.ipmi => "IPMI Interface",
|
||||
.sercos => "SERCOS Interface (IEC 61491)",
|
||||
.canbus => "CANbus Controller",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const usb = struct {
|
||||
pub const ProgIf = enum(u8) {
|
||||
uhci = 0x00,
|
||||
ohci = 0x10,
|
||||
ehci = 0x20,
|
||||
xhci = 0x30,
|
||||
unspecified = 0x80,
|
||||
device = 0xFE,
|
||||
|
||||
pub fn name(self: ProgIf) []const u8 {
|
||||
return switch (self) {
|
||||
.uhci => "UHCI Controller",
|
||||
.ohci => "OHCI Controller",
|
||||
.ehci => "EHCI (USB2) Controller",
|
||||
.xhci => "XHCI (USB3) Controller",
|
||||
.unspecified => "Unspecified",
|
||||
.device => "USB Device (not a host controller)",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
pub const wireless = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
irda = 0x00,
|
||||
consumer_ir = 0x01,
|
||||
rf = 0x10,
|
||||
bluetooth = 0x11,
|
||||
broadband = 0x12,
|
||||
ethernet_802_1a = 0x20,
|
||||
ethernet_802_1b = 0x21,
|
||||
|
||||
pub fn name(self: SubClass) []const u8 {
|
||||
return switch (self) {
|
||||
.irda => "iRDA Compatible Controller",
|
||||
.consumer_ir => "Consumer IR Controller",
|
||||
.rf => "RF Controller",
|
||||
.bluetooth => "Bluetooth Controller",
|
||||
.broadband => "Broadband Controller",
|
||||
.ethernet_802_1a => "Ethernet Controller (802.1a)",
|
||||
.ethernet_802_1b => "Ethernet Controller (802.1b)",
|
||||
};
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
// --- Raw-byte decoding (what a function reports in its header) -------------------------
|
||||
|
||||
/// The name of an exhaustive class-code enum member, or null if `value` is not one — the
|
||||
/// bridge from a raw config byte to a named taxonomy above.
|
||||
fn enumName(comptime Enum: type, value: u8) ?[]const u8 {
|
||||
return (std.enums.fromInt(Enum, value) orelse return null).name();
|
||||
}
|
||||
|
||||
/// Name of the base class (byte 0x0B), e.g. `0x06` -> "Bridge".
|
||||
pub fn className(base: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x00 => "Unclassified",
|
||||
0x01 => "Mass Storage Controller",
|
||||
0x02 => "Network Controller",
|
||||
0x03 => "Display Controller",
|
||||
0x04 => "Multimedia Controller",
|
||||
0x05 => "Memory Controller",
|
||||
0x06 => "Bridge",
|
||||
0x07 => "Simple Communication Controller",
|
||||
0x08 => "Base System Peripheral",
|
||||
0x09 => "Input Device Controller",
|
||||
0x0A => "Docking Station",
|
||||
0x0B => "Processor",
|
||||
0x0C => "Serial Bus Controller",
|
||||
0x0D => "Wireless Controller",
|
||||
0x0E => "Intelligent Controller",
|
||||
0x0F => "Satellite Communication Controller",
|
||||
0x10 => "Encryption Controller",
|
||||
0x11 => "Signal Processing Controller",
|
||||
0x12 => "Processing Accelerator",
|
||||
0x13 => "Non-Essential Instrumentation",
|
||||
0x40 => "Co-Processor",
|
||||
0xFF => "Unassigned Class (Vendor specific)",
|
||||
else => "Unknown",
|
||||
};
|
||||
return @as(BaseClass, @enumFromInt(base)).name();
|
||||
}
|
||||
|
||||
/// Name of the subclass within its base class, e.g. `(0x06, 0x01)` -> "ISA Bridge".
|
||||
/// Subclass `0x80` is "Other" by PCI convention; anything unlisted is "Unknown".
|
||||
pub fn subclassName(base: u8, subclass: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x01 => switch (subclass) {
|
||||
0x00 => "SCSI Bus Controller",
|
||||
0x01 => "IDE Controller",
|
||||
0x02 => "Floppy Disk Controller",
|
||||
0x03 => "IPI Bus Controller",
|
||||
0x04 => "RAID Controller",
|
||||
0x05 => "ATA Controller",
|
||||
0x06 => "Serial ATA Controller",
|
||||
0x07 => "Serial Attached SCSI Controller",
|
||||
0x08 => "Non-Volatile Memory Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x02 => switch (subclass) {
|
||||
0x00 => "Ethernet Controller",
|
||||
0x01 => "Token Ring Controller",
|
||||
0x02 => "FDDI Controller",
|
||||
0x03 => "ATM Controller",
|
||||
0x04 => "ISDN Controller",
|
||||
0x06 => "PICMG 2.14 Multi Computing Controller",
|
||||
0x07 => "Infiniband Controller",
|
||||
0x08 => "Fabric Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x03 => switch (subclass) {
|
||||
0x00 => "VGA Compatible Controller",
|
||||
0x01 => "XGA Controller",
|
||||
0x02 => "3D Controller (Not VGA-Compatible)",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x04 => switch (subclass) {
|
||||
0x00 => "Multimedia Video Controller",
|
||||
0x01 => "Multimedia Audio Controller",
|
||||
0x02 => "Computer Telephony Device",
|
||||
0x03 => "Audio Device",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x05 => switch (subclass) {
|
||||
0x00 => "RAM Controller",
|
||||
0x01 => "Flash Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x06 => switch (subclass) {
|
||||
0x00 => "Host Bridge",
|
||||
0x01 => "ISA Bridge",
|
||||
0x02 => "EISA Bridge",
|
||||
0x03 => "MCA Bridge",
|
||||
0x04 => "PCI-to-PCI Bridge",
|
||||
0x05 => "PCMCIA Bridge",
|
||||
0x06 => "NuBus Bridge",
|
||||
0x07 => "CardBus Bridge",
|
||||
0x08 => "RACEway Bridge",
|
||||
0x09 => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||
0x0A => "InfiniBand-to-PCI Host Bridge",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x07 => switch (subclass) {
|
||||
0x00 => "Serial Controller",
|
||||
0x01 => "Parallel Controller",
|
||||
0x02 => "Multiport Serial Controller",
|
||||
0x03 => "Modem",
|
||||
0x04 => "IEEE 488.1/2 (GPIB) Controller",
|
||||
0x05 => "Smart Card Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x08 => switch (subclass) {
|
||||
0x00 => "PIC",
|
||||
0x01 => "DMA Controller",
|
||||
0x02 => "Timer",
|
||||
0x03 => "RTC Controller",
|
||||
0x04 => "PCI Hot-Plug Controller",
|
||||
0x05 => "SD Host Controller",
|
||||
0x06 => "IOMMU",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x09 => switch (subclass) {
|
||||
0x00 => "Keyboard Controller",
|
||||
0x01 => "Digitizer Pen",
|
||||
0x02 => "Mouse Controller",
|
||||
0x03 => "Scanner Controller",
|
||||
0x04 => "Gameport Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x0C => switch (subclass) {
|
||||
0x00 => "FireWire (IEEE 1394) Controller",
|
||||
0x01 => "ACCESS Bus Controller",
|
||||
0x02 => "SSA",
|
||||
0x03 => "USB Controller",
|
||||
0x04 => "Fibre Channel",
|
||||
0x05 => "SMBus Controller",
|
||||
0x06 => "InfiniBand Controller",
|
||||
0x07 => "IPMI Interface",
|
||||
0x08 => "SERCOS Interface (IEC 61491)",
|
||||
0x09 => "CANbus Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x0D => switch (subclass) {
|
||||
0x00 => "iRDA Compatible Controller",
|
||||
0x01 => "Consumer IR Controller",
|
||||
0x10 => "RF Controller",
|
||||
0x11 => "Bluetooth Controller",
|
||||
0x12 => "Broadband Controller",
|
||||
0x20 => "Ethernet Controller (802.1a)",
|
||||
0x21 => "Ethernet Controller (802.1b)",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
else => defaultSubclass(subclass),
|
||||
const named: ?[]const u8 = switch (@as(BaseClass, @enumFromInt(base))) {
|
||||
.mass_storage => enumName(mass_storage.SubClass, subclass),
|
||||
.network => enumName(network.SubClass, subclass),
|
||||
.display => enumName(display.SubClass, subclass),
|
||||
.multimedia => enumName(multimedia.SubClass, subclass),
|
||||
.memory => enumName(memory.SubClass, subclass),
|
||||
.bridge => enumName(bridge.SubClass, subclass),
|
||||
.simple_communication => enumName(simple_communication.SubClass, subclass),
|
||||
.base_system_peripheral => enumName(base_system_peripheral.SubClass, subclass),
|
||||
.input_device => enumName(input_device.SubClass, subclass),
|
||||
.serial_bus => enumName(serial_bus.SubClass, subclass),
|
||||
.wireless => enumName(wireless.SubClass, subclass),
|
||||
else => null,
|
||||
};
|
||||
return named orelse defaultSubclass(subclass);
|
||||
}
|
||||
|
||||
fn defaultSubclass(subclass: u8) []const u8 {
|
||||
@@ -175,68 +502,34 @@ fn defaultSubclass(subclass: u8) []const u8 {
|
||||
/// Returns "" when the prog-IF carries no standard meaning for this class/subclass —
|
||||
/// callers just print the hex byte in that case.
|
||||
pub fn progIfName(base: u8, subclass: u8, prog_if: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x01 => switch (subclass) {
|
||||
0x06 => switch (prog_if) { // Serial ATA
|
||||
0x00 => "Vendor Specific Interface",
|
||||
0x01 => "AHCI 1.0",
|
||||
0x02 => "Serial Storage Bus",
|
||||
else => "",
|
||||
},
|
||||
0x08 => switch (prog_if) { // Non-Volatile Memory
|
||||
0x01 => "NVMHCI",
|
||||
0x02 => "NVM Express",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
const named: ?[]const u8 = switch (@as(BaseClass, @enumFromInt(base))) {
|
||||
.mass_storage => switch (std.enums.fromInt(mass_storage.SubClass, subclass) orelse return "") {
|
||||
.serial_ata => enumName(mass_storage.serial_ata.ProgIf, prog_if),
|
||||
.non_volatile_memory => enumName(mass_storage.non_volatile_memory.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x03 => switch (subclass) {
|
||||
0x00 => switch (prog_if) { // VGA Compatible
|
||||
0x00 => "VGA Controller",
|
||||
0x01 => "8514-Compatible Controller",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.display => switch (std.enums.fromInt(display.SubClass, subclass) orelse return "") {
|
||||
.vga_compatible => enumName(display.vga_compatible.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x06 => switch (subclass) {
|
||||
0x04 => switch (prog_if) { // PCI-to-PCI Bridge
|
||||
0x00 => "Normal Decode",
|
||||
0x01 => "Subtractive Decode",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.bridge => switch (std.enums.fromInt(bridge.SubClass, subclass) orelse return "") {
|
||||
.pci_to_pci => enumName(bridge.pci_to_pci.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x07 => switch (subclass) {
|
||||
0x00 => switch (prog_if) { // Serial Controller
|
||||
0x00 => "8250-Compatible (Generic XT)",
|
||||
0x01 => "16450-Compatible",
|
||||
0x02 => "16550-Compatible",
|
||||
0x03 => "16650-Compatible",
|
||||
0x04 => "16750-Compatible",
|
||||
0x05 => "16850-Compatible",
|
||||
0x06 => "16950-Compatible",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.simple_communication => switch (std.enums.fromInt(simple_communication.SubClass, subclass) orelse return "") {
|
||||
.serial => enumName(simple_communication.serial.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
0x0C => switch (subclass) {
|
||||
0x03 => switch (prog_if) { // USB Controller
|
||||
0x00 => "UHCI Controller",
|
||||
0x10 => "OHCI Controller",
|
||||
0x20 => "EHCI (USB2) Controller",
|
||||
0x30 => "XHCI (USB3) Controller",
|
||||
0x80 => "Unspecified",
|
||||
0xFE => "USB Device (not a host controller)",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
.serial_bus => switch (std.enums.fromInt(serial_bus.SubClass, subclass) orelse return "") {
|
||||
.usb => enumName(serial_bus.usb.ProgIf, prog_if),
|
||||
else => null,
|
||||
},
|
||||
else => "",
|
||||
else => null,
|
||||
};
|
||||
return named orelse "";
|
||||
}
|
||||
|
||||
test "decodes the common class codes" {
|
||||
const std = @import("std");
|
||||
const eq = std.testing.expectEqualStrings;
|
||||
|
||||
const isa = ClassCode.unpack(0x06_01_00);
|
||||
@@ -251,11 +544,25 @@ test "decodes the common class codes" {
|
||||
try eq("AHCI 1.0", progIfName(ahci.base, ahci.subclass, ahci.prog_if));
|
||||
|
||||
const xhci = ClassCode.unpack(0x0C_03_30);
|
||||
try eq("Serial Bus Controller", className(xhci.base));
|
||||
try eq("USB Controller", subclassName(xhci.base, xhci.subclass));
|
||||
try eq("XHCI (USB3) Controller", progIfName(xhci.base, xhci.subclass, xhci.prog_if));
|
||||
}
|
||||
|
||||
// Unknowns and the "Other" convention.
|
||||
try eq("Other", subclassName(0x02, 0x80));
|
||||
try eq("Unknown", subclassName(0x06, 0x7E));
|
||||
try eq("", progIfName(0x06, 0x00, 0x00)); // host bridge: prog-IF has no standard name
|
||||
test "unlisted codes fall back without a wrong name" {
|
||||
const eq = std.testing.expectEqualStrings;
|
||||
try eq("Unknown", className(0x77)); // no such base class
|
||||
try eq("Other", subclassName(0x01, 0x80)); // 0x80 is the PCI "Other" convention
|
||||
try eq("Unknown", subclassName(0x01, 0x7A)); // unlisted mass-storage subclass
|
||||
try eq("", progIfName(0x01, 0x06, 0x7F)); // no standard SATA prog-IF for 0x7F
|
||||
try eq("", progIfName(0x02, 0x00, 0x00)); // class with no prog-IF taxonomy at all
|
||||
}
|
||||
|
||||
test "named parts pack to the raw triple" {
|
||||
const xhci = ClassCode{
|
||||
.base = @intFromEnum(BaseClass.serial_bus),
|
||||
.subclass = @intFromEnum(serial_bus.SubClass.usb),
|
||||
.prog_if = @intFromEnum(serial_bus.usb.ProgIf.xhci),
|
||||
};
|
||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||
}
|
||||
|
||||
@@ -22,13 +22,14 @@ pub const Resource = device_model.Resource;
|
||||
pub const ResourceKind = device_model.ResourceKind;
|
||||
pub const Hal = device_model.Hal;
|
||||
pub const PowerInformation = acpi.PowerInformation;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInformation = acpi.PlatformInformation;
|
||||
pub const RegisterAccess = acpi.RegisterAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||
/// owned by the ring-3 acpi service), so they are not here.
|
||||
pub fn powerInformation() PowerInformation {
|
||||
return acpi.power_information;
|
||||
}
|
||||
@@ -39,18 +40,6 @@ pub fn platformInformation() PlatformInformation {
|
||||
return acpi.platform_information;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
||||
/// count against this.
|
||||
pub fn amlDeviceCount() usize {
|
||||
return acpi.amlDeviceCount();
|
||||
}
|
||||
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// The usable logical processors discovered during enumeration — one entry per
|
||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||
@@ -92,12 +81,8 @@ pub fn discover(
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls. Soft-off
|
||||
/// (S5) is not a kernel operation — the ring-3 acpi service owns it (docs/power.md).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
|
||||
+10
-63
@@ -1,34 +1,17 @@
|
||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
||||
//! Machine reboot: restart via the FADT reset register, with legacy fallbacks.
|
||||
//!
|
||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
||||
//!
|
||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
||||
//! re-initialisation, a milestone of its own.
|
||||
//! Built on the register map `acpi` extracted from the FADT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Soft-off (ACPI S5) and suspend (S3) are
|
||||
//! **not** here: they need the AML sleep-state (`_Sx`) values, which the kernel no
|
||||
//! longer parses — the ring-3 acpi service owns power management (it re-parses the
|
||||
//! blobs and writes the PM1 control register itself). See docs/power.md. Reboot
|
||||
//! stays in the kernel because it needs no AML — only the FADT reset register and
|
||||
//! the well-known legacy fallbacks — so it survives as a last-resort restart.
|
||||
|
||||
const acpi = @import("acpi.zig");
|
||||
const device_model = @import("device-model.zig");
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||
|
||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||
pub fn enable(hal: Hal) void {
|
||||
const pi = acpi.power_information;
|
||||
if (!pi.pm1a_cnt.present()) return;
|
||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||
|
||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||
var spins: usize = 0;
|
||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
}
|
||||
|
||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
@@ -48,42 +31,6 @@ pub fn reboot(hal: Hal) void {
|
||||
delay();
|
||||
}
|
||||
|
||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
enable(hal);
|
||||
const pi = acpi.power_information;
|
||||
const s5 = pi.s5 orelse return;
|
||||
|
||||
if (pi.pm1a_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
}
|
||||
if (pi.pm1b_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
}
|
||||
delay();
|
||||
}
|
||||
|
||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
||||
_ = hal;
|
||||
return error.Unsupported;
|
||||
}
|
||||
|
||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
||||
/// [12:10], SLP_EN in bit 13.
|
||||
fn sleepValue(slp_typ: u8) u32 {
|
||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||
}
|
||||
|
||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
return p.*;
|
||||
}
|
||||
return hal.pioRead(register.width, @intCast(register.address));
|
||||
}
|
||||
|
||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
@@ -93,8 +40,8 @@ fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// A short busy-wait so a reset takes effect before we fall through to the next
|
||||
/// method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// from being optimised away.
|
||||
fn delay() void {
|
||||
var i: usize = 0;
|
||||
|
||||
+139
-51
@@ -8,7 +8,7 @@
|
||||
//! buffer at any offset, and bitmap bytes are packed structs so no caller ever needs a magic
|
||||
//! mask. Class, subclass, and protocol code tables live in usb-ids.zig.
|
||||
|
||||
const DeviceState = enum(u8) {
|
||||
pub const DeviceState = enum(u8) {
|
||||
// Immediately after the USB device is attached to the USB system, it is in this state.
|
||||
// The USB specifications do not define the state of a USB device that is detached from
|
||||
// a USB system.
|
||||
@@ -47,7 +47,7 @@ const DeviceState = enum(u8) {
|
||||
suspended,
|
||||
};
|
||||
|
||||
const RequestCode = enum(u8) {
|
||||
pub const RequestCode = enum(u8) {
|
||||
get_status = 0,
|
||||
clear_feature = 1,
|
||||
set_feature = 3,
|
||||
@@ -59,10 +59,15 @@ const RequestCode = enum(u8) {
|
||||
get_interface = 10,
|
||||
set_interface = 11,
|
||||
sync_frame = 12,
|
||||
// Non-exhaustive: class-specific requests (HID, mass storage) reuse this byte
|
||||
// field with codes from their own class's namespace — see the class-request
|
||||
// constructors below. Some class codes numerically coincide with a standard
|
||||
// one; the wire byte is what matters, and the constructors set it explicitly.
|
||||
_,
|
||||
};
|
||||
|
||||
// Direction of an endpoint, from the host's point of view
|
||||
const EndpointDirection = enum(u1) {
|
||||
pub const EndpointDirection = enum(u1) {
|
||||
out = 0,
|
||||
in = 1,
|
||||
};
|
||||
@@ -74,7 +79,7 @@ const EndpointDirection = enum(u1) {
|
||||
|
||||
// The bus address of a device, assigned by the host with SET_ADDRESS. Addresses are 7 bits
|
||||
// wide.
|
||||
const DeviceAddress = enum(u7) {
|
||||
pub const DeviceAddress = enum(u7) {
|
||||
// The default address every device answers at after a reset, until SET_ADDRESS
|
||||
// completes
|
||||
default = 0,
|
||||
@@ -82,7 +87,7 @@ const DeviceAddress = enum(u7) {
|
||||
};
|
||||
|
||||
// Identifies a configuration; from ConfigurationDescriptor.configuration_value.
|
||||
const ConfigurationValue = enum(u8) {
|
||||
pub const ConfigurationValue = enum(u8) {
|
||||
// Not configured: returned by GET_CONFIGURATION while the device is in the address
|
||||
// state, and passed to SET_CONFIGURATION to return a configured device to the address
|
||||
// state
|
||||
@@ -92,11 +97,11 @@ const ConfigurationValue = enum(u8) {
|
||||
|
||||
// Identifies an interface within a configuration; from
|
||||
// InterfaceDescriptor.interface_number.
|
||||
const InterfaceNumber = enum(u8) { _ };
|
||||
pub const InterfaceNumber = enum(u8) { _ };
|
||||
|
||||
// Selects between the alternate settings of one interface; from
|
||||
// InterfaceDescriptor.alternate_setting.
|
||||
const AlternateSetting = enum(u8) {
|
||||
pub const AlternateSetting = enum(u8) {
|
||||
// The default setting of an interface
|
||||
default = 0,
|
||||
_,
|
||||
@@ -104,7 +109,7 @@ const AlternateSetting = enum(u8) {
|
||||
|
||||
// The number of an endpoint within a device, 4 bits wide. The direction bit carried
|
||||
// alongside it tells the two endpoints sharing a number apart.
|
||||
const EndpointNumber = enum(u4) {
|
||||
pub const EndpointNumber = enum(u4) {
|
||||
// Endpoint zero: the default control pipe every device provides
|
||||
default_control = 0,
|
||||
_,
|
||||
@@ -112,7 +117,7 @@ const EndpointNumber = enum(u4) {
|
||||
|
||||
// Index of a STRING descriptor, stored in descriptors that reference a string and passed to
|
||||
// GET_DESCRIPTOR to read it.
|
||||
const StringIndex = enum(u8) {
|
||||
pub const StringIndex = enum(u8) {
|
||||
// The device has no string descriptor for this field
|
||||
none = 0,
|
||||
_,
|
||||
@@ -121,7 +126,7 @@ const StringIndex = enum(u8) {
|
||||
// Characteristics of a device request (the bmRequestType field of a set-up packet). Fields are
|
||||
// declared least-significant first: recipient occupies bits 4...0, kind bits 6...5, and
|
||||
// direction bit 7.
|
||||
const RequestType = packed struct(u8) {
|
||||
pub const RequestType = packed struct(u8) {
|
||||
// The recipient of the request (values 4...31 are reserved)
|
||||
recipient: Recipient,
|
||||
// The type of the request
|
||||
@@ -129,27 +134,27 @@ const RequestType = packed struct(u8) {
|
||||
// Data transfer direction. The value of this bit is ignored when length is zero.
|
||||
direction: Direction,
|
||||
|
||||
const Recipient = enum(u5) {
|
||||
pub const Recipient = enum(u5) {
|
||||
device = 0,
|
||||
interface = 1,
|
||||
endpoint = 2,
|
||||
other = 3,
|
||||
};
|
||||
|
||||
const Kind = enum(u2) {
|
||||
pub const Kind = enum(u2) {
|
||||
standard = 0,
|
||||
class = 1,
|
||||
vendor = 2,
|
||||
reserved = 3,
|
||||
};
|
||||
|
||||
const Direction = enum(u1) {
|
||||
pub const Direction = enum(u1) {
|
||||
host_to_device = 0,
|
||||
device_to_host = 1,
|
||||
};
|
||||
};
|
||||
|
||||
const Request = extern struct {
|
||||
pub const Request = extern struct {
|
||||
// Characteristics of the request
|
||||
request_type: RequestType,
|
||||
// Specific request
|
||||
@@ -175,7 +180,7 @@ const Request = extern struct {
|
||||
// The format of the index field when request_type specifies an endpoint as the
|
||||
// recipient. The host should always set the direction bit to zero (but the device
|
||||
// should accept either value) when the endpoint is part of a control pipe.
|
||||
const EndpointIndex = packed struct(u16) {
|
||||
pub const EndpointIndex = packed struct(u16) {
|
||||
// Endpoint number
|
||||
number: EndpointNumber,
|
||||
// Reserved (reset to zero)
|
||||
@@ -188,7 +193,7 @@ const Request = extern struct {
|
||||
|
||||
// The format of the index field when request_type specifies an interface as the
|
||||
// recipient.
|
||||
const InterfaceIndex = packed struct(u16) {
|
||||
pub const InterfaceIndex = packed struct(u16) {
|
||||
// Interface number
|
||||
number: u8,
|
||||
// Reserved (reset to zero)
|
||||
@@ -199,7 +204,7 @@ const Request = extern struct {
|
||||
// descriptor type in the high byte, and the descriptor index in the low byte. The index
|
||||
// is used to select a specific descriptor (only for CONFIGURATION and STRING
|
||||
// descriptors) when several descriptors of that type are implemented by a device.
|
||||
const DescriptorValue = packed struct(u16) {
|
||||
pub const DescriptorValue = packed struct(u16) {
|
||||
// Descriptor index
|
||||
index: u8 = 0,
|
||||
// Descriptor type
|
||||
@@ -209,7 +214,7 @@ const Request = extern struct {
|
||||
|
||||
// Feature selectors, used as the value field of CLEAR_FEATURE and SET_FEATURE requests. The
|
||||
// comment on each value notes the recipient the selector applies to.
|
||||
const FeatureSelector = enum(u16) {
|
||||
pub const FeatureSelector = enum(u16) {
|
||||
// Halts an endpoint (recipient: endpoint)
|
||||
endpoint_halt = 0,
|
||||
// Enables or disables the device's remote wakeup capability (recipient: device)
|
||||
@@ -223,7 +228,7 @@ const FeatureSelector = enum(u16) {
|
||||
// with the test_mode feature selector. Values 06h...3Fh are reserved for standard test
|
||||
// selectors and C0h...FFh for vendor-specific test modes; all other unlisted values are
|
||||
// reserved.
|
||||
const TestMode = enum(u8) {
|
||||
pub const TestMode = enum(u8) {
|
||||
test_j = 0x01,
|
||||
test_k = 0x02,
|
||||
test_se0_nak = 0x03,
|
||||
@@ -234,7 +239,7 @@ const TestMode = enum(u8) {
|
||||
|
||||
// The two bytes returned by a GET_STATUS request directed at a device. Fields are declared
|
||||
// least-significant first.
|
||||
const DeviceStatus = packed struct(u16) {
|
||||
pub const DeviceStatus = packed struct(u16) {
|
||||
// Whether the device is currently self-powered (as opposed to bus-powered). This bit
|
||||
// cannot be changed with the SET_FEATURE or CLEAR_FEATURE requests.
|
||||
self_powered: bool,
|
||||
@@ -248,7 +253,7 @@ const DeviceStatus = packed struct(u16) {
|
||||
|
||||
// The two bytes returned by a GET_STATUS request directed at an endpoint. (A GET_STATUS
|
||||
// request directed at an interface returns two bytes that are entirely reserved.)
|
||||
const EndpointStatus = packed struct(u16) {
|
||||
pub const EndpointStatus = packed struct(u16) {
|
||||
// Whether the endpoint is currently halted. Set with the SET_FEATURE request using the
|
||||
// endpoint_halt feature selector, and cleared with CLEAR_FEATURE.
|
||||
halted: bool,
|
||||
@@ -258,7 +263,7 @@ const EndpointStatus = packed struct(u16) {
|
||||
|
||||
// A target for the standard requests that may be directed at the device, an interface, or
|
||||
// an endpoint.
|
||||
const Target = union(enum) {
|
||||
pub const Target = union(enum) {
|
||||
device,
|
||||
interface: InterfaceNumber,
|
||||
endpoint: Request.EndpointIndex,
|
||||
@@ -287,7 +292,7 @@ const Target = union(enum) {
|
||||
// Reads the status of the given target: bit-cast the two bytes the device returns into a
|
||||
// DeviceStatus or an EndpointStatus. (The two bytes returned for an interface are entirely
|
||||
// reserved.)
|
||||
fn getStatus(target: Target) Request {
|
||||
pub fn getStatus(target: Target) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = target.recipient(),
|
||||
@@ -303,7 +308,7 @@ fn getStatus(target: Target) Request {
|
||||
|
||||
// Clears or disables the given feature. A device cannot be taken out of a test mode with
|
||||
// this request; test_mode is only cleared by cycling power.
|
||||
fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
||||
pub fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = target.recipient(),
|
||||
@@ -319,7 +324,7 @@ fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
||||
|
||||
// Sets or enables the given feature. For the test_mode feature selector, use setTestMode
|
||||
// instead: the test selector rides in the high byte of the index field.
|
||||
fn setFeature(feature: FeatureSelector, target: Target) Request {
|
||||
pub fn setFeature(feature: FeatureSelector, target: Target) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = target.recipient(),
|
||||
@@ -335,7 +340,7 @@ fn setFeature(feature: FeatureSelector, target: Target) Request {
|
||||
|
||||
// Puts a hi-speed device into the given test mode: a SET_FEATURE request with the test_mode
|
||||
// feature selector and the test selector in the high byte of the index field.
|
||||
fn setTestMode(mode: TestMode) Request {
|
||||
pub fn setTestMode(mode: TestMode) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
@@ -352,7 +357,7 @@ fn setTestMode(mode: TestMode) Request {
|
||||
// Assigns the device its bus address, moving it from the default state to the address
|
||||
// state. The device does not answer at the new address until the status stage of this
|
||||
// request completes.
|
||||
fn setAddress(address: DeviceAddress) Request {
|
||||
pub fn setAddress(address: DeviceAddress) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
@@ -372,7 +377,7 @@ fn setAddress(address: DeviceAddress) Request {
|
||||
// - language_id selects the language of a string descriptor, and is zero otherwise.
|
||||
// - length is the number of bytes to read; a device never returns more than length bytes,
|
||||
// but may return less if the descriptor is shorter.
|
||||
fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||
pub fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
@@ -389,7 +394,7 @@ fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, l
|
||||
// Updates an existing descriptor or adds a new one (optional; many devices do not support
|
||||
// this request). The parameters mirror getDescriptor; the descriptor itself is sent in the
|
||||
// DATA stage.
|
||||
fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||
pub fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
@@ -405,7 +410,7 @@ fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, l
|
||||
|
||||
// Reads the currently active configuration: @enumFromInt the byte the device returns into a
|
||||
// ConfigurationValue, which is none while the device is not configured.
|
||||
fn getConfiguration() Request {
|
||||
pub fn getConfiguration() Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
@@ -422,7 +427,7 @@ fn getConfiguration() Request {
|
||||
// Selects the configuration with the given configuration_value (from
|
||||
// ConfigurationDescriptor.configuration_value), moving the device from the address state to
|
||||
// the configured state. Selecting none returns the device to the address state.
|
||||
fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
||||
pub fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
@@ -438,7 +443,7 @@ fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
||||
|
||||
// Reads the alternate setting currently selected for the given interface: @enumFromInt the
|
||||
// byte the device returns into an AlternateSetting.
|
||||
fn getInterface(interface: InterfaceNumber) Request {
|
||||
pub fn getInterface(interface: InterfaceNumber) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .interface,
|
||||
@@ -454,7 +459,7 @@ fn getInterface(interface: InterfaceNumber) Request {
|
||||
|
||||
// Selects an alternate setting (from InterfaceDescriptor.alternate_setting) for the given
|
||||
// interface.
|
||||
fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting) Request {
|
||||
pub fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .interface,
|
||||
@@ -470,7 +475,7 @@ fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting)
|
||||
|
||||
// Reads the two-byte number of the frame in which the given isochronous endpoint's
|
||||
// repeating pattern of transfers begins.
|
||||
fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
||||
pub fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .endpoint,
|
||||
@@ -484,7 +489,79 @@ fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
||||
};
|
||||
}
|
||||
|
||||
const DescriptorType = enum(u8) {
|
||||
// Class-specific requests. These carry a `kind = .class` request_type and a
|
||||
// request_code from the interface's class namespace (not the standard
|
||||
// RequestCode set above); the code is written into the same byte field, which
|
||||
// is why RequestCode is non-exhaustive. Each is directed at an interface, whose
|
||||
// number rides in the index field.
|
||||
|
||||
// The HID class request codes (USB HID 1.11 §7.2). Only the ones danos issues
|
||||
// are named; the field on the wire is the raw byte.
|
||||
pub const HidRequestCode = enum(u8) {
|
||||
get_report = 0x01,
|
||||
get_idle = 0x02,
|
||||
get_protocol = 0x03,
|
||||
set_report = 0x09,
|
||||
set_idle = 0x0A,
|
||||
set_protocol = 0x0B,
|
||||
};
|
||||
|
||||
// The two protocols a boot-capable HID device can run (USB HID 1.11 §7.2.5).
|
||||
// A driver selects `boot` for the simplified fixed-format boot report, usable
|
||||
// before a full report-descriptor parser exists.
|
||||
pub const HidProtocol = enum(u8) {
|
||||
boot = 0,
|
||||
report = 1,
|
||||
};
|
||||
|
||||
// SET_PROTOCOL: choose the boot or report protocol on a HID interface.
|
||||
pub fn setProtocol(interface: InterfaceNumber, protocol: HidProtocol) Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .host_to_device },
|
||||
.request_code = @enumFromInt(@intFromEnum(HidRequestCode.set_protocol)),
|
||||
.value = @intFromEnum(protocol),
|
||||
.index = @intFromEnum(interface),
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// SET_IDLE: bound a HID interface's report rate. `duration` is in 4 ms units
|
||||
// (0 means report only on change); `report_id` selects a report (0 = all).
|
||||
pub fn setIdle(interface: InterfaceNumber, duration: u8, report_id: u8) Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .host_to_device },
|
||||
.request_code = @enumFromInt(@intFromEnum(HidRequestCode.set_idle)),
|
||||
.value = (@as(u16, duration) << 8) | report_id,
|
||||
.index = @intFromEnum(interface),
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Bulk-Only Mass Storage Reset (USB MSC BOT §3.1): ready a mass-storage
|
||||
// interface for the next Command Block Wrapper after a protocol error.
|
||||
pub fn bulkOnlyMassStorageReset(interface: InterfaceNumber) Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .host_to_device },
|
||||
.request_code = @enumFromInt(0xFF),
|
||||
.value = 0,
|
||||
.index = @intFromEnum(interface),
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Get Max LUN (USB MSC BOT §3.2): read the highest logical unit number the
|
||||
// device supports (0 for a single-LUN flash drive). One byte is returned.
|
||||
pub fn getMaxLun(interface: InterfaceNumber) Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .interface, .kind = .class, .direction = .device_to_host },
|
||||
.request_code = @enumFromInt(0xFE),
|
||||
.value = 0,
|
||||
.index = @intFromEnum(interface),
|
||||
.length = 1,
|
||||
};
|
||||
}
|
||||
|
||||
pub const DescriptorType = enum(u8) {
|
||||
device = 1,
|
||||
configuration = 2,
|
||||
string = 3,
|
||||
@@ -496,7 +573,7 @@ const DescriptorType = enum(u8) {
|
||||
_,
|
||||
};
|
||||
|
||||
const DeviceDescriptor = extern struct {
|
||||
pub const DeviceDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// DEVICE Descriptor Type
|
||||
@@ -541,7 +618,7 @@ const DeviceDescriptor = extern struct {
|
||||
configuration_count: u8,
|
||||
};
|
||||
|
||||
const DeviceQualifierDescriptor = extern struct {
|
||||
pub const DeviceQualifierDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// DEVICE_QUALIFIER Descriptor Type
|
||||
@@ -564,7 +641,7 @@ const DeviceQualifierDescriptor = extern struct {
|
||||
reserved: u8,
|
||||
};
|
||||
|
||||
const ConfigurationDescriptor = extern struct {
|
||||
pub const ConfigurationDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// CONFIGURATION Descriptor Type
|
||||
@@ -593,7 +670,7 @@ const ConfigurationDescriptor = extern struct {
|
||||
max_power: u8,
|
||||
|
||||
// Configuration characteristics. Fields are declared least-significant first.
|
||||
const Attributes = packed struct(u8) {
|
||||
pub const Attributes = packed struct(u8) {
|
||||
// Reserved, reset to zero (D4...0)
|
||||
reserved: u5,
|
||||
// Whether Remote Wakeup is supported by this configuration (D5)
|
||||
@@ -612,9 +689,9 @@ const ConfigurationDescriptor = extern struct {
|
||||
// its alternative speed. The structure of the OTHER_SPEED_CONFIGURATION is identical to that
|
||||
// of the CONFIGURATION descriptor; the only difference is that the descriptor_type field
|
||||
// reflects that the descriptor is an OTHER_SPEED_CONFIGURATION descriptor.
|
||||
const OtherSpeedConfigurationDescriptor = ConfigurationDescriptor;
|
||||
pub const OtherSpeedConfigurationDescriptor = ConfigurationDescriptor;
|
||||
|
||||
const InterfaceDescriptor = extern struct {
|
||||
pub const InterfaceDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// INTERFACE Descriptor Type
|
||||
@@ -654,7 +731,7 @@ const InterfaceDescriptor = extern struct {
|
||||
interface_index: StringIndex,
|
||||
};
|
||||
|
||||
const EndpointDescriptor = extern struct {
|
||||
pub const EndpointDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// ENDPOINT Descriptor Type
|
||||
@@ -683,7 +760,7 @@ const EndpointDescriptor = extern struct {
|
||||
interval: u8,
|
||||
|
||||
// The address of an endpoint. Fields are declared least-significant first.
|
||||
const Address = packed struct(u8) {
|
||||
pub const Address = packed struct(u8) {
|
||||
// Endpoint Number (D3...0)
|
||||
number: EndpointNumber,
|
||||
// Reserved, reset to zero (D6...4)
|
||||
@@ -693,7 +770,7 @@ const EndpointDescriptor = extern struct {
|
||||
};
|
||||
|
||||
// An endpoint's attributes. Fields are declared least-significant first.
|
||||
const Attributes = packed struct(u8) {
|
||||
pub const Attributes = packed struct(u8) {
|
||||
// Transfer Type (D1...0)
|
||||
transfer_type: TransferType,
|
||||
// Synchronization Type; isochronous endpoints only, reserved and reset to zero for
|
||||
@@ -706,21 +783,21 @@ const EndpointDescriptor = extern struct {
|
||||
reserved: u2,
|
||||
};
|
||||
|
||||
const TransferType = enum(u2) {
|
||||
pub const TransferType = enum(u2) {
|
||||
control = 0,
|
||||
isochronous = 1,
|
||||
bulk = 2,
|
||||
interrupt = 3,
|
||||
};
|
||||
|
||||
const Synchronization = enum(u2) {
|
||||
pub const Synchronization = enum(u2) {
|
||||
none = 0,
|
||||
asynchronous = 1,
|
||||
adaptive = 2,
|
||||
synchronous = 3,
|
||||
};
|
||||
|
||||
const Usage = enum(u2) {
|
||||
pub const Usage = enum(u2) {
|
||||
data = 0,
|
||||
feedback = 1,
|
||||
implicit_feedback_data = 2,
|
||||
@@ -728,7 +805,7 @@ const EndpointDescriptor = extern struct {
|
||||
};
|
||||
|
||||
// The maximum packet size of an endpoint. Fields are declared least-significant first.
|
||||
const MaxPacketSize = packed struct(u16) {
|
||||
pub const MaxPacketSize = packed struct(u16) {
|
||||
// Maximum packet size in bytes (bits 10...0)
|
||||
size: u11,
|
||||
// Number of additional transaction opportunities per microframe, for high-speed
|
||||
@@ -739,7 +816,7 @@ const EndpointDescriptor = extern struct {
|
||||
reserved: u3,
|
||||
};
|
||||
|
||||
const AdditionalTransactions = enum(u2) {
|
||||
pub const AdditionalTransactions = enum(u2) {
|
||||
// None (1 transaction per microframe)
|
||||
none = 0,
|
||||
// 1 additional (2 transactions per microframe)
|
||||
@@ -755,7 +832,7 @@ const EndpointDescriptor = extern struct {
|
||||
// header, followed by the variable-length payload:
|
||||
// - index 0: an array of two-byte LANGID codes (wLangID[0] through wLangID[x])
|
||||
// - other indices: a Unicode string of N bytes
|
||||
const StringDescriptor = extern struct {
|
||||
pub const StringDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// STRING Descriptor Type
|
||||
@@ -834,7 +911,7 @@ test "bitmap packings match the specification" {
|
||||
try expect(hid_type != .device);
|
||||
}
|
||||
|
||||
fn expectRequestBytes(request: Request, expected: [8]u8) !void {
|
||||
pub fn expectRequestBytes(request: Request, expected: [8]u8) !void {
|
||||
try std.testing.expectEqualSlices(u8, &expected, std.mem.asBytes(&request));
|
||||
}
|
||||
|
||||
@@ -855,3 +932,14 @@ test "standard request constructors encode the specification's set-up packets" {
|
||||
try expectRequestBytes(setInterface(@enumFromInt(2), @enumFromInt(1)), .{ 0x01, 11, 1, 0, 2, 0, 0, 0 });
|
||||
try expectRequestBytes(syncFrame(.{ .number = @enumFromInt(3), .direction = .in }), .{ 0x82, 12, 0, 0, 0x83, 0, 2, 0 });
|
||||
}
|
||||
|
||||
test "class request constructors encode the specification's set-up packets" {
|
||||
// bmRequestType for a host-to-device class request to an interface = 0x21;
|
||||
// device-to-host = 0xA1. The request_code byte is the class code, not a
|
||||
// standard one — SET_PROTOCOL 0x0B, SET_IDLE 0x0A, BOT reset 0xFF, Max LUN 0xFE.
|
||||
try expectRequestBytes(setProtocol(@enumFromInt(0), .boot), .{ 0x21, 0x0B, 0, 0, 0, 0, 0, 0 });
|
||||
try expectRequestBytes(setProtocol(@enumFromInt(1), .report), .{ 0x21, 0x0B, 1, 0, 1, 0, 0, 0 });
|
||||
try expectRequestBytes(setIdle(@enumFromInt(1), 0, 0), .{ 0x21, 0x0A, 0, 0, 1, 0, 0, 0 });
|
||||
try expectRequestBytes(bulkOnlyMassStorageReset(@enumFromInt(0)), .{ 0x21, 0xFF, 0, 0, 0, 0, 0, 0 });
|
||||
try expectRequestBytes(getMaxLun(@enumFromInt(0)), .{ 0xA1, 0xFE, 0, 0, 0, 0, 1, 0 });
|
||||
}
|
||||
|
||||
+62
-19
@@ -11,7 +11,7 @@
|
||||
|
||||
// Base class codes (assigned by the USB-IF). The comment on each value notes where the code
|
||||
// may legally appear: in the device descriptor, in interface descriptors, or both.
|
||||
const Class = enum(u8) {
|
||||
pub const Class = enum(u8) {
|
||||
// Use class information in the interface descriptors (device descriptor only). Each
|
||||
// interface within a configuration specifies its own class information and the various
|
||||
// interfaces operate independently.
|
||||
@@ -72,8 +72,8 @@ const Class = enum(u8) {
|
||||
|
||||
// Subclass and protocol codes qualified by Class.hub. Hubs have no subclass codes; the
|
||||
// protocol distinguishes the hub's transaction-translator arrangement.
|
||||
const hub = struct {
|
||||
const Protocol = enum(u8) {
|
||||
pub const hub = struct {
|
||||
pub const Protocol = enum(u8) {
|
||||
// Full-speed hub
|
||||
full_speed = 0x00,
|
||||
// Hi-speed hub with a single transaction translator
|
||||
@@ -87,8 +87,8 @@ const hub = struct {
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.hid.
|
||||
const hid = struct {
|
||||
const SubClass = enum(u8) {
|
||||
pub const hid = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
// No subclass
|
||||
none = 0x00,
|
||||
// Boot interface: the device also supports the simplified boot protocol, usable by
|
||||
@@ -98,7 +98,7 @@ const hid = struct {
|
||||
};
|
||||
|
||||
// Only meaningful when the subclass is boot
|
||||
const Protocol = enum(u8) {
|
||||
pub const Protocol = enum(u8) {
|
||||
none = 0x00,
|
||||
keyboard = 0x01,
|
||||
mouse = 0x02,
|
||||
@@ -109,8 +109,8 @@ const hid = struct {
|
||||
// Subclass and protocol codes qualified by Class.mass_storage. The subclass identifies the
|
||||
// command set the device understands; the protocol identifies the transport used to carry
|
||||
// commands, data, and status over the bus.
|
||||
const mass_storage = struct {
|
||||
const SubClass = enum(u8) {
|
||||
pub const mass_storage = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
// SCSI command set not reported; de facto, treat as scsi
|
||||
not_reported = 0x00,
|
||||
// Reduced Block Commands: typically flash devices
|
||||
@@ -134,7 +134,7 @@ const mass_storage = struct {
|
||||
_,
|
||||
};
|
||||
|
||||
const Protocol = enum(u8) {
|
||||
pub const Protocol = enum(u8) {
|
||||
// Control/Bulk/Interrupt with command completion interrupt
|
||||
cbi_completion_interrupt = 0x00,
|
||||
// Control/Bulk/Interrupt without command completion interrupt
|
||||
@@ -152,8 +152,8 @@ const mass_storage = struct {
|
||||
// Subclass and protocol codes qualified by Class.communications (CDC). The protocol codes
|
||||
// are model-specific; the useful invariant is the subclass, which selects the control model
|
||||
// the interface implements.
|
||||
const communications = struct {
|
||||
const SubClass = enum(u8) {
|
||||
pub const communications = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
// Direct line control model
|
||||
direct_line = 0x01,
|
||||
// Abstract control model: USB modems and serial adapters
|
||||
@@ -185,15 +185,15 @@ const communications = struct {
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.wireless_controller.
|
||||
const wireless_controller = struct {
|
||||
const SubClass = enum(u8) {
|
||||
pub const wireless_controller = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
// Radio frequency controllers
|
||||
radio_frequency = 0x01,
|
||||
_,
|
||||
};
|
||||
|
||||
// Only meaningful when the subclass is radio_frequency
|
||||
const Protocol = enum(u8) {
|
||||
pub const Protocol = enum(u8) {
|
||||
// Bluetooth programming interface
|
||||
bluetooth = 0x01,
|
||||
// Ultra-wideband radio control
|
||||
@@ -207,15 +207,15 @@ const wireless_controller = struct {
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.miscellaneous.
|
||||
const miscellaneous = struct {
|
||||
const SubClass = enum(u8) {
|
||||
pub const miscellaneous = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
// Common class
|
||||
common = 0x02,
|
||||
_,
|
||||
};
|
||||
|
||||
// Only meaningful when the subclass is common
|
||||
const Protocol = enum(u8) {
|
||||
pub const Protocol = enum(u8) {
|
||||
// Interface association descriptor: at the device level, announces that the
|
||||
// configuration groups interfaces into functions with IADs
|
||||
interface_association = 0x01,
|
||||
@@ -224,8 +224,8 @@ const miscellaneous = struct {
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.application_specific.
|
||||
const application_specific = struct {
|
||||
const SubClass = enum(u8) {
|
||||
pub const application_specific = struct {
|
||||
pub const SubClass = enum(u8) {
|
||||
// Device firmware upgrade
|
||||
firmware_upgrade = 0x01,
|
||||
// IrDA bridge
|
||||
@@ -236,6 +236,23 @@ const application_specific = struct {
|
||||
};
|
||||
};
|
||||
|
||||
/// Pack a (class, subclass, protocol) triple into one 0xCCSSPP value — the
|
||||
/// bus-native identity a USB bus driver reports in `ChildAdded.identity` and the
|
||||
/// device manager matches on (the USB analog of a packed PCI class code). Mirrors
|
||||
/// `pci_class.ClassCode.pack`, so both sides build/decode the identical u64.
|
||||
pub fn packTriple(class: u8, subclass: u8, protocol: u8) u64 {
|
||||
return (@as(u64, class) << 16) | (@as(u64, subclass) << 8) | protocol;
|
||||
}
|
||||
|
||||
/// The inverse of `packTriple`.
|
||||
pub fn unpackTriple(triple: u64) struct { class: u8, subclass: u8, protocol: u8 } {
|
||||
return .{
|
||||
.class = @truncate(triple >> 16),
|
||||
.subclass = @truncate(triple >> 8),
|
||||
.protocol = @truncate(triple),
|
||||
};
|
||||
}
|
||||
|
||||
test "class codes match the USB-IF assignments" {
|
||||
const std = @import("std");
|
||||
const expectEqual = std.testing.expectEqual;
|
||||
@@ -262,3 +279,29 @@ test "class codes match the USB-IF assignments" {
|
||||
_ = miscellaneous.Protocol.interface_association;
|
||||
_ = application_specific.SubClass.firmware_upgrade;
|
||||
}
|
||||
|
||||
test "packTriple / unpackTriple round-trip the identity a bus driver reports" {
|
||||
const std = @import("std");
|
||||
const expectEqual = std.testing.expectEqual;
|
||||
|
||||
// A boot keyboard interface: HID / boot / keyboard.
|
||||
const keyboard = packTriple(
|
||||
@intFromEnum(Class.hid),
|
||||
@intFromEnum(hid.SubClass.boot),
|
||||
@intFromEnum(hid.Protocol.keyboard),
|
||||
);
|
||||
try expectEqual(@as(u64, 0x03_01_01), keyboard);
|
||||
|
||||
// A flash drive interface: mass storage / SCSI / bulk-only.
|
||||
const storage = packTriple(
|
||||
@intFromEnum(Class.mass_storage),
|
||||
@intFromEnum(mass_storage.SubClass.scsi),
|
||||
@intFromEnum(mass_storage.Protocol.bulk_only),
|
||||
);
|
||||
try expectEqual(@as(u64, 0x08_06_50), storage);
|
||||
|
||||
const parts = unpackTriple(storage);
|
||||
try expectEqual(@as(u8, 0x08), parts.class);
|
||||
try expectEqual(@as(u8, 0x06), parts.subclass);
|
||||
try expectEqual(@as(u8, 0x50), parts.protocol);
|
||||
}
|
||||
|
||||
@@ -1,211 +0,0 @@
|
||||
//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one.
|
||||
//!
|
||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
||||
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
||||
//!
|
||||
//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children
|
||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
||||
//! every one of those resources is contained in what `bus` was granted. A comparator
|
||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
||||
//!
|
||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
||||
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
||||
//! arbitrary physical memory.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
|
||||
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
||||
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
||||
return .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = hpet_base + 0x100 + 0x20 * n,
|
||||
.len = 0x20,
|
||||
};
|
||||
}
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
||||
for (0..d.resource_count) |j| {
|
||||
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The parent's MMIO resource, and its IRQ if it has one.
|
||||
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
||||
var mmio: device.ResourceDescriptor = undefined;
|
||||
var irq: ?device.ResourceDescriptor = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
const r = d.resources[j];
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
||||
}
|
||||
return .{ .mmio = mmio, .irq = irq };
|
||||
}
|
||||
|
||||
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent == parent_id) return d.id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("bus: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const parent = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("bus: no HPET\n");
|
||||
return;
|
||||
};
|
||||
const resource = resourcesOf(parent);
|
||||
|
||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
||||
//
|
||||
// Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk
|
||||
// binary — so hpet may own the HPET already. That's not an error, it's the
|
||||
// capability model working: exit quietly and leave the device to its owner. The
|
||||
// `bus` test spawns bus alone, so there it wins the claim.
|
||||
if (!device.claim(parent.id)) {
|
||||
_ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Enumerate the bus: ask the hardware how many children it has.
|
||||
const base = device.mmioMap(parent.id, 0) orelse {
|
||||
_ = runtime.system.write("bus: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
||||
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
||||
|
||||
// Publish one child per comparator, each owning only its own window.
|
||||
var published: u64 = 0;
|
||||
var n: u64 = 0;
|
||||
while (n < n_children) : (n += 1) {
|
||||
var child = std.mem.zeroes(device.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device.DeviceClass.timer);
|
||||
child.pci_class = device.no_pci_class;
|
||||
child.hid_len = 6;
|
||||
child.hid[0..6].* = "hpet-t".*;
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = timerWindow(resource.mmio.start, n);
|
||||
// Comparators share the block's interrupt line; only one child can bind it,
|
||||
// but all of them may legitimately name it.
|
||||
if (resource.irq) |i| {
|
||||
child.resources[child.resource_count] = i;
|
||||
child.resource_count += 1;
|
||||
}
|
||||
|
||||
if (device.register(parent.id, &child) == null) {
|
||||
_ = runtime.system.write("bus: register failed\n");
|
||||
return;
|
||||
}
|
||||
published += 1;
|
||||
}
|
||||
|
||||
// The negative case. A window one byte past the end of the parent's must be
|
||||
// refused — otherwise device_register would be "map any physical page you like".
|
||||
// Confirm the table did not grow, not merely that the call returned null: null
|
||||
// also means NoSpace/BadParent, so a size check is what actually proves the
|
||||
// *containment* rule fired.
|
||||
const before = device.enumerate(buffer);
|
||||
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
||||
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
||||
rogue.resource_count = 1;
|
||||
rogue.resources[0] = .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = resource.mmio.start + resource.mmio.len,
|
||||
.len = 0x1000,
|
||||
};
|
||||
if (device.register(parent.id, &rogue) != null) {
|
||||
_ = runtime.system.write("bus: FAIL out-of-window child was accepted\n");
|
||||
return;
|
||||
}
|
||||
if (device.enumerate(buffer) != before) {
|
||||
_ = runtime.system.write("bus: FAIL rogue child leaked into the table\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// And confirm the children came back with the right parent and a *narrower*
|
||||
// window than the bus — read from the table, not from our own memory.
|
||||
const total = device.enumerate(buffer);
|
||||
var seen: u64 = 0;
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent != parent.id) continue;
|
||||
const w = d.resources[0];
|
||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
||||
_ = runtime.system.write("bus: FAIL child window is not inside the bus\n");
|
||||
return;
|
||||
}
|
||||
seen += 1;
|
||||
}
|
||||
if (seen != published) {
|
||||
_ = runtime.system.write("bus: FAIL child count mismatch\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
||||
// a different process; here bus plays both parts, which exercises the same path.
|
||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
||||
//
|
||||
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
||||
// The *resource* is narrow even though the page isn't.)
|
||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
||||
_ = runtime.system.write("bus: FAIL no child to claim\n");
|
||||
return;
|
||||
};
|
||||
if (!device.claim(child_id)) {
|
||||
_ = runtime.system.write("bus: FAIL could not claim own child\n");
|
||||
return;
|
||||
}
|
||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
||||
_ = runtime.system.write("bus: FAIL child mmio_map refused\n");
|
||||
return;
|
||||
};
|
||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
||||
if (via_child.* != via_bus.*) {
|
||||
_ = runtime.system.write("bus: FAIL child window does not alias the bus register\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
||||
// kernel. Grab a page, free it, and register through the stale address: if the
|
||||
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
||||
// would triple-fault QEMU and the test would time out instead of printing ok.
|
||||
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
||||
if (!runtime.system.mmapFailed(scratch)) {
|
||||
_ = runtime.system.munmap(scratch, 0x1000);
|
||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
||||
if (device.register(parent.id, descriptor) != null) {
|
||||
_ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("bus: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -1,195 +0,0 @@
|
||||
//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to
|
||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
||||
//!
|
||||
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
||||
//! scheduler queue; the core runs other work or idles. That is the point of the
|
||||
//! exercise — a driver is a process that sleeps until its device has something to
|
||||
//! say (see docs/drivers.md).
|
||||
//!
|
||||
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
||||
//! simpler, but level is the discipline every real device line needs, and it forces
|
||||
//! the full cycle to be correct:
|
||||
//!
|
||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
||||
//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||
//! hpet irq_ack -> kernel unmasks the GSI
|
||||
//!
|
||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
||||
//!
|
||||
//! Register map (HPET spec 1.0a):
|
||||
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
||||
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
||||
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
||||
//! 0x0F0 MAIN_COUNTER
|
||||
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
||||
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
||||
//! 0x108 TIMER0_COMPARATOR
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const mmio = @import("mmio");
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
const register_general_configuration = 0x010;
|
||||
const register_int_status = 0x020;
|
||||
const register_main_counter = 0x0F0;
|
||||
const register_timer0_configuration = 0x100;
|
||||
const register_timer0_comparator = 0x108;
|
||||
|
||||
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
||||
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
||||
const tn_int_type_level: u64 = 1 << 1;
|
||||
const tn_int_enb: u64 = 1 << 2;
|
||||
const tn_type_periodic: u64 = 1 << 3;
|
||||
const tn_route_shift = 9;
|
||||
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
||||
|
||||
/// Interrupts to observe before declaring victory.
|
||||
const target_ticks = 5;
|
||||
|
||||
/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio).
|
||||
/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so
|
||||
/// UC writes are already ordered) — no barriers are needed here; the point is the
|
||||
/// typed, arch-portable access every driver should use.
|
||||
inline fn rd(base: usize, off: usize) u64 {
|
||||
return mmio.read(u64, base + off);
|
||||
}
|
||||
inline fn wr(base: usize, off: usize, value: u64) void {
|
||||
mmio.write(u64, base + off, value);
|
||||
}
|
||||
|
||||
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
||||
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
||||
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
// Skip comparator children a bus driver may have published below the block
|
||||
// (see system/drivers/bus/bus.zig) — we want the register block itself.
|
||||
if (d.parent != device.no_parent) continue;
|
||||
var mmio_index: ?u64 = null;
|
||||
var irq: ?u64 = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
switch (d.resources[j].kind) {
|
||||
@intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j,
|
||||
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
if (mmio_index) |m| if (irq) |i| {
|
||||
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
||||
_ = runtime.system.write("hpet: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const hpet = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("hpet: no HPET with an IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(hpet.device_id)) {
|
||||
_ = runtime.system.write("hpet: claim failed\n");
|
||||
return;
|
||||
}
|
||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
||||
_ = runtime.system.write("hpet: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
||||
// to raise exactly this line — the kernel will only bind the one it recorded.
|
||||
const gsi = hpet.gsi;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("hpet: create_ipc_endpoint failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// --- program the hardware ------------------------------------------------
|
||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
||||
const femtos_per_tick = rd(base, register_general_cap) >> 32;
|
||||
if (femtos_per_tick == 0) {
|
||||
_ = runtime.system.write("hpet: bad HPET period\n");
|
||||
return;
|
||||
}
|
||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
||||
|
||||
// Stop the counter and take the legacy route off while we reconfigure.
|
||||
wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt));
|
||||
|
||||
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
||||
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
||||
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
||||
// timer driver does anyway.
|
||||
var t0 = rd(base, register_timer0_configuration);
|
||||
t0 &= ~(tn_route_mask | tn_type_periodic);
|
||||
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
||||
wr(base, register_timer0_configuration, t0);
|
||||
|
||||
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
||||
wr(base, register_int_status, 1);
|
||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
||||
wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable);
|
||||
|
||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
||||
_ = runtime.system.write("hpet: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("hpet: bound, sleeping until the hardware speaks\n");
|
||||
|
||||
// --- the driver loop -----------------------------------------------------
|
||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
||||
// runs only because an interrupt fired.
|
||||
var receive: [64]u8 = undefined;
|
||||
var count: usize = 0;
|
||||
while (count < target_ticks) {
|
||||
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
||||
// next line runs only because the HPET raised its line.
|
||||
const r = ipc.replyWait(endpoint, &.{}, &receive, null);
|
||||
if (!r.isNotification()) continue; // a client request, not our IRQ
|
||||
|
||||
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
||||
// line is still asserted and unmasking would refire immediately.
|
||||
wr(base, register_int_status, 1);
|
||||
count += 1;
|
||||
|
||||
if (count < target_ticks) {
|
||||
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
|
||||
} else {
|
||||
// Last one: stop the source rather than re-arming, so the line is left
|
||||
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
||||
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
||||
// the line again a moment later.
|
||||
wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb);
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpet: irq\n");
|
||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
||||
_ = runtime.system.write("hpet: irq_ack failed\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpet: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -1,12 +1,12 @@
|
||||
//! /system/drivers/pci-bus — the PCI bus driver: enumeration moved out of ring 0
|
||||
//! (docs/m19-m20-plan.md, M19). The device manager matches the `pci_host_bridge`
|
||||
//! (docs/discovery.md). The device manager matches the `pci_host_bridge`
|
||||
//! node and spawns one instance per bridge, the bridge's device id as argv[1] —
|
||||
//! the same per-device contract as usb-xhci-bus.
|
||||
//!
|
||||
//! M19.1 (this increment): claim the bridge, map its ECAM window (resource 0;
|
||||
//! the bus range and the MMIO apertures follow it), walk every
|
||||
//! bus/device/function config header, and log what the walk finds — ending
|
||||
//! with "pci-bus: N functions found", which the `pci-scan` scenario compares
|
||||
//! with "/system/drivers/pci-bus: N functions found", which the `pci-scan` scenario compares
|
||||
//! against the kernel's own enumeration. Registration and reports (M19.2), and
|
||||
//! the kernel walk's retirement (M19.3), build on this proven-equivalent scan.
|
||||
|
||||
@@ -14,12 +14,29 @@ const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Log a discovered function with its (class / subclass / prog-IF) triple decoded
|
||||
/// to human names — the boot-log breadcrumb that says *what* the hardware is, so
|
||||
/// "class 0x01 (Mass Storage Controller) subclass 0x06 (Serial ATA Controller)
|
||||
/// progif 0x01 (AHCI 1.0)" reads straight off the log when writing a new driver.
|
||||
/// A dedicated wider buffer than `writeLine`'s, since the decoded names are long.
|
||||
fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
||||
const cc = pci_class.ClassCode.unpack(@truncate(class_triple));
|
||||
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||
var line: [200]u8 = undefined;
|
||||
const text = if (pif.len != 0)
|
||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||
_ = runtime.system.write(text);
|
||||
}
|
||||
|
||||
var bridge_id: u64 = protocol.no_device;
|
||||
var ecam_base: usize = 0;
|
||||
var ecam_physical: u64 = 0;
|
||||
@@ -57,37 +74,37 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(bridge_id)) {
|
||||
writeLine("pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||
writeLine("/system/drivers/pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||
return false;
|
||||
}
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("pci-bus: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: out of memory\n");
|
||||
return false;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == bridge_id) break d;
|
||||
} else {
|
||||
writeLine("pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||
writeLine("/system/drivers/pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||
return false;
|
||||
};
|
||||
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
||||
// range rides beside it. The MMIO apertures (M19.0) come after both.
|
||||
if (descriptor.resource_count < 2 or descriptor.resources[0].kind != @intFromEnum(device.ResourceKind.memory)) {
|
||||
_ = runtime.system.write("pci-bus: bridge has no ECAM window\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: bridge has no ECAM window\n");
|
||||
return false;
|
||||
}
|
||||
const bus_range = for (descriptor.resources[0..@intCast(descriptor.resource_count)]) |resource| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.bus_range)) break resource;
|
||||
} else {
|
||||
_ = runtime.system.write("pci-bus: bridge has no bus range\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: bridge has no bus range\n");
|
||||
return false;
|
||||
};
|
||||
start_bus = bus_range.start;
|
||||
bus_count = bus_range.len;
|
||||
ecam_physical = descriptor.resources[0].start;
|
||||
ecam_base = device.mmioMap(bridge_id, 0) orelse {
|
||||
_ = runtime.system.write("pci-bus: ECAM mmio_map failed\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: ECAM mmio_map failed\n");
|
||||
return false;
|
||||
};
|
||||
|
||||
@@ -99,17 +116,17 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
if (manager == null) runtime.system.sleep(20);
|
||||
}
|
||||
const h = manager orelse {
|
||||
_ = runtime.system.write("pci-bus: no device manager to hello\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: no device manager to hello\n");
|
||||
return false;
|
||||
};
|
||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = bridge_id };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||
_ = runtime.system.write("pci-bus: hello call failed\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: hello call failed\n");
|
||||
return false;
|
||||
};
|
||||
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||
_ = runtime.system.write("pci-bus: hello refused\n");
|
||||
_ = runtime.system.write("/system/drivers/pci-bus: hello refused\n");
|
||||
return false;
|
||||
}
|
||||
manager_handle = h;
|
||||
@@ -137,12 +154,12 @@ fn scan() void {
|
||||
if (vendor_device & 0xFFFF == 0xFFFF) continue;
|
||||
const class_revision = configRead(bus, dev, function, 0x08);
|
||||
found += 1;
|
||||
writeLine("pci-bus: {d}:{d}.{d} class 0x{x:0>6}\n", .{ bus, dev, function, class_revision >> 8 });
|
||||
logFunction(bus, dev, function, class_revision >> 8);
|
||||
registerAndReport(bus, dev, function, class_revision >> 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
writeLine("pci-bus: {d} functions found\n", .{found});
|
||||
writeLine("/system/drivers/pci-bus: {d} functions found\n", .{found});
|
||||
}
|
||||
|
||||
/// Register one function under the bridge and report it to the manager. The
|
||||
@@ -212,7 +229,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
}
|
||||
|
||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||
writeLine("pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||
writeLine("/system/drivers/pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||
return;
|
||||
};
|
||||
const report = protocol.ChildAdded{
|
||||
@@ -223,7 +240,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||
writeLine("/system/drivers/pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||
};
|
||||
}
|
||||
|
||||
@@ -238,7 +255,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||
writeLine("/system/drivers/pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
|
||||
@@ -72,17 +72,17 @@ fn modifierWord(modifiers: scancode.ModifierSnapshot) u32 {
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const hid = init.arguments.get(1).?;
|
||||
if (hid.len == 0) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: no HID argument\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("system/drivers/ps2-bus/keyboard: starting for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: starting for hid {s}\n", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("system/drivers/ps2-bus/keyboard: no device for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: no device for hid {s}\n", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -90,38 +90,38 @@ pub fn main(init: runtime.process.Init) void {
|
||||
// absent (as today) it defaults to us.
|
||||
const layout_name = init.arguments.get(2) orelse "us";
|
||||
const layout = xkb.byName(layout_name) orelse xkb.us;
|
||||
writeLine("system/drivers/ps2-bus/keyboard: layout {s}\n", .{layout.name});
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: layout {s}\n", .{layout.name});
|
||||
|
||||
// Attach to the bus: hand it our endpoint, and it forwards every byte the
|
||||
// keyboard sends (it owns the controller; we own the decoding).
|
||||
const bus = lookupBus() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: ps2-bus service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: ps2-bus service unavailable\n");
|
||||
return;
|
||||
};
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: no endpoint\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
var attach = ps2.AttachRequest{ .device_type = @intFromEnum(ps2.DeviceType.keyboard) };
|
||||
var attach_reply: [@sizeOf(ps2.AttachReply)]u8 = undefined;
|
||||
const attached = ipc.callCap(bus, std.mem.asBytes(&attach), &attach_reply, endpoint) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: attach call failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: attach call failed\n");
|
||||
return;
|
||||
};
|
||||
if (attached.len < @sizeOf(ps2.AttachReply) or
|
||||
std.mem.bytesToValue(ps2.AttachReply, attach_reply[0..@sizeOf(ps2.AttachReply)]).status != @intFromEnum(ps2.AttachStatus.ok))
|
||||
{
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: attach refused\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: attach refused\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Broadcast keyboard events through the input service so programs can listen
|
||||
// for them (docs/input.md).
|
||||
var source = runtime.input.connectSource() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: input service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/keyboard: ok\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: ok\n");
|
||||
|
||||
var decoder = scancode.Decoder{};
|
||||
var state = scancode.KeyboardState{};
|
||||
|
||||
@@ -51,50 +51,50 @@ pub fn main(init: runtime.process.Init) void {
|
||||
const hid = init.arguments.get(1).?;
|
||||
|
||||
if (hid.len == 0) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: no HID argument\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("system/drivers/ps2-bus/mouse: starting for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/mouse: starting for hid {s}\n", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
writeLine("/system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
// Attach to the bus: hand it our endpoint, and it forwards every byte the
|
||||
// mouse sends (it owns the controller; we own the decoding).
|
||||
const bus = lookupBus() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: ps2-bus service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: ps2-bus service unavailable\n");
|
||||
return;
|
||||
};
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: no endpoint\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
var attach = ps2.AttachRequest{ .device_type = @intFromEnum(ps2.DeviceType.mouse) };
|
||||
var attach_reply: [@sizeOf(ps2.AttachReply)]u8 = undefined;
|
||||
const attached = ipc.callCap(bus, std.mem.asBytes(&attach), &attach_reply, endpoint) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: attach call failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: attach call failed\n");
|
||||
return;
|
||||
};
|
||||
if (attached.len < @sizeOf(ps2.AttachReply) or
|
||||
std.mem.bytesToValue(ps2.AttachReply, attach_reply[0..@sizeOf(ps2.AttachReply)]).status != @intFromEnum(ps2.AttachStatus.ok))
|
||||
{
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: attach refused\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: attach refused\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Broadcast mouse events through the input service so programs can listen
|
||||
// for them (docs/input.md).
|
||||
var source = runtime.input.connectSource() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: input service unavailable\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.system.write("system/drivers/ps2-bus/mouse: ok\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: ok\n");
|
||||
|
||||
var assembler = mouse_packet.Assembler{};
|
||||
var buttons: u32 = 0;
|
||||
|
||||
@@ -30,19 +30,19 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
/// attaches, or null if nothing was spawned.
|
||||
fn spawnIdentifiedDriver(controller: ps2.Controller, port: ps2.Port) ?ps2.DeviceType {
|
||||
const device_type = controller.identifyDevice(port) orelse {
|
||||
writeLine("system/drivers/ps2-bus: identify timed out on port {s}\n", .{@tagName(port)});
|
||||
writeLine("/system/drivers/ps2-bus: identify timed out on port {s}\n", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const driver_name = device_type.driverName() orelse {
|
||||
writeLine("system/drivers/ps2-bus: unrecognized device on port {s}\n", .{@tagName(port)});
|
||||
writeLine("/system/drivers/ps2-bus: unrecognized device on port {s}\n", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const hid = device_type.hid() orelse "";
|
||||
if (runtime.system.spawnWithArguments(driver_name, &.{hid}) != null) {
|
||||
writeLine("system/drivers/ps2-bus: port {s} is a {s}, spawned {s}\n", .{ @tagName(port), hid, driver_name });
|
||||
writeLine("/system/drivers/ps2-bus: port {s} is a {s}, spawned {s}\n", .{ @tagName(port), hid, driver_name });
|
||||
return device_type;
|
||||
}
|
||||
writeLine("system/drivers/ps2-bus: failed to spawn {s}\n", .{driver_name});
|
||||
writeLine("/system/drivers/ps2-bus: failed to spawn {s}\n", .{driver_name});
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -83,7 +83,7 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
const device_type = maybe_type orelse continue;
|
||||
if (@intFromEnum(device_type) != request.device_type) continue;
|
||||
port_endpoints[port_index] = endpoint;
|
||||
writeLine("system/drivers/ps2-bus: {s} driver attached\n", .{@tagName(device_type)});
|
||||
writeLine("/system/drivers/ps2-bus: {s} driver attached\n", .{@tagName(device_type)});
|
||||
return reply.write(out, .ok);
|
||||
}
|
||||
return reply.write(out, .no_such_device);
|
||||
@@ -91,7 +91,7 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -103,16 +103,16 @@ pub fn main() void {
|
||||
// is on which port is decided later by identify, not by this HID.
|
||||
const maybe_controller_device_descriptor = device.findDeviceDescriptorByHid(buffer, acpi_ids.HardwareId.ps2_keyboard.hid());
|
||||
if (maybe_controller_device_descriptor) |controller_device_descriptor| {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: found PS/2 controller\n");
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: initializing controller\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: found PS/2 controller\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: initializing controller\n");
|
||||
|
||||
if (!device.claim(controller_device_descriptor.id)) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: unable to claim controller \n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: unable to claim controller \n");
|
||||
return;
|
||||
}
|
||||
|
||||
const controller = ps2.Controller.init(controller_device_descriptor) orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller is missing its IO ports\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller is missing its IO ports\n");
|
||||
return;
|
||||
};
|
||||
maybe_controller = controller;
|
||||
@@ -123,7 +123,7 @@ pub fn main() void {
|
||||
controller.flushOutputBuffer();
|
||||
|
||||
const current = controller.readConfigurationByte() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -132,49 +132,49 @@ pub fn main() void {
|
||||
ps2.configuration_first_port_translation);
|
||||
|
||||
if (controller.writeConfigurationByte(update) == null) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
return;
|
||||
}
|
||||
|
||||
if (controller.performSelfTest()) |reply| {
|
||||
if (reply != ps2.response_controller_test_passed) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: perform controller self test failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: perform controller self test failed\n");
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller self test timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller self test timed out\n");
|
||||
return;
|
||||
}
|
||||
|
||||
has_two_channels = controller.hasTwoChannels() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller channels timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller channels timed out\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (has_two_channels) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: has two channels\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: has two channels\n");
|
||||
// keep the bus quiet until we have tested the ports and are ready to use them
|
||||
controller.disablePort(.two);
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: has one channel\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: has one channel\n");
|
||||
}
|
||||
|
||||
// interface tests: always test port 1, test port 2 only if it exists
|
||||
const port_one_works = (controller.testPort(.one) orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 1 test timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 1 test timed out\n");
|
||||
return;
|
||||
}) == ps2.response_port_test_passed;
|
||||
|
||||
var port_two_works = false;
|
||||
if (has_two_channels) {
|
||||
port_two_works = (controller.testPort(.two) orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 2 test timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 2 test timed out\n");
|
||||
return;
|
||||
}) == ps2.response_port_test_passed;
|
||||
}
|
||||
|
||||
if (!port_one_works and !port_two_works) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: no usable ports\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: no usable ports\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -188,16 +188,16 @@ pub fn main() void {
|
||||
// abort bring-up of the other one
|
||||
if (port_one_works) {
|
||||
if (controller.resetDevice(.one)) |passed| {
|
||||
if (!passed) _ = runtime.system.write("system/drivers/ps2-bus: port 1 device reset failed\n");
|
||||
if (!passed) _ = runtime.system.write("/system/drivers/ps2-bus: port 1 device reset failed\n");
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 1 device reset timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 1 device reset timed out\n");
|
||||
}
|
||||
}
|
||||
if (port_two_works) {
|
||||
if (controller.resetDevice(.two)) |passed| {
|
||||
if (!passed) _ = runtime.system.write("system/drivers/ps2-bus: port 2 device reset failed\n");
|
||||
if (!passed) _ = runtime.system.write("/system/drivers/ps2-bus: port 2 device reset failed\n");
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: port 2 device reset timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: port 2 device reset timed out\n");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -207,13 +207,13 @@ pub fn main() void {
|
||||
if (port_one_works) port_device_types[@intFromEnum(ps2.Port.one)] = spawnIdentifiedDriver(controller, .one);
|
||||
if (port_two_works) port_device_types[@intFromEnum(ps2.Port.two)] = spawnIdentifiedDriver(controller, .two);
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: no PS/2 controller found\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: no PS/2 controller found\n");
|
||||
return;
|
||||
}
|
||||
|
||||
const controller = maybe_controller.?;
|
||||
const interrupt_index = maybe_interrupt_index orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller is missing its IRQ\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller is missing its IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -221,11 +221,11 @@ pub fn main() void {
|
||||
// well-known id so the children can find it, the way input subscribers find
|
||||
// the input service.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: no endpoint\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
if (!ipc.register(.ps2_bus, endpoint)) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: register failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: register failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -234,7 +234,7 @@ pub fn main() void {
|
||||
// let the controller raise them — an interrupt with nobody bound is lost.
|
||||
controller.drainOutputBuffer();
|
||||
if (!device.irqBind(controller.device_id, interrupt_index, endpoint)) {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: irq_bind failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -253,21 +253,21 @@ pub fn main() void {
|
||||
.gsi = descriptor.resources[auxiliary_index].start,
|
||||
};
|
||||
} else {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: auxiliary irq_bind failed\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: auxiliary irq_bind failed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var configuration = controller.readConfigurationByte() orelse {
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: controller configuration timed out\n");
|
||||
return;
|
||||
};
|
||||
if (port_device_types[@intFromEnum(ps2.Port.one)] != null) configuration |= ps2.Port.one.interruptBit();
|
||||
if (maybe_auxiliary_interrupt != null) configuration |= ps2.Port.two.interruptBit();
|
||||
_ = controller.writeConfigurationByte(configuration);
|
||||
|
||||
_ = runtime.system.write("system/drivers/ps2-bus: ok\n");
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus: ok\n");
|
||||
|
||||
// The forwarding loop: an IRQ1 notification drains the output buffer, routing
|
||||
// each byte to the attached driver of the port it came from; a client message
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
//! Pure decoders for USB HID **boot-protocol** reports — the simplified,
|
||||
//! fixed-format reports a boot keyboard and boot mouse send, the USB analog of
|
||||
//! the PS/2 scancode and mouse-packet decoders. No I/O: these turn report bytes
|
||||
//! into make/break transitions and motion, which the usb-hid drivers publish to
|
||||
//! the input service. Host-testable in isolation (like mouse-packet.zig).
|
||||
//!
|
||||
//! "Boot protocol" is a USB HID term (USB HID 1.11 §B) — the device reports in
|
||||
//! this fixed layout after SET_PROTOCOL(boot); it has nothing to do with system
|
||||
//! boot.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// --- keyboard ---------------------------------------------------------------
|
||||
|
||||
/// The 8-byte boot keyboard report: a modifier bitmap, a reserved byte, and up
|
||||
/// to six concurrently-pressed key usages.
|
||||
pub const KeyboardReport = extern struct {
|
||||
modifiers: u8 = 0,
|
||||
reserved: u8 = 0,
|
||||
keys: [6]u8 = .{ 0, 0, 0, 0, 0, 0 },
|
||||
};
|
||||
|
||||
// The modifier byte's bits (HID keyboard boot report).
|
||||
pub const modifier_left_control: u8 = 1 << 0;
|
||||
pub const modifier_left_shift: u8 = 1 << 1;
|
||||
pub const modifier_left_alt: u8 = 1 << 2;
|
||||
pub const modifier_left_gui: u8 = 1 << 3;
|
||||
pub const modifier_right_control: u8 = 1 << 4;
|
||||
pub const modifier_right_shift: u8 = 1 << 5;
|
||||
pub const modifier_right_alt: u8 = 1 << 6;
|
||||
pub const modifier_right_gui: u8 = 1 << 7;
|
||||
|
||||
pub const TransitionKind = enum { pressed, released };
|
||||
|
||||
/// One key going down or up. `usage` is a HID keyboard-page usage — modifier keys
|
||||
/// map to usages 224..231 — which is exactly the input protocol's `Keycode`.
|
||||
pub const Transition = struct { kind: TransitionKind, usage: u8 };
|
||||
|
||||
// A report can change at most all 8 modifiers and all 6 keys at once.
|
||||
pub const max_transitions = 8 + 6;
|
||||
|
||||
pub const Transitions = struct {
|
||||
items: [max_transitions]Transition = undefined,
|
||||
count: usize = 0,
|
||||
|
||||
fn add(self: *Transitions, transition: Transition) void {
|
||||
if (self.count < self.items.len) {
|
||||
self.items[self.count] = transition;
|
||||
self.count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
pub fn slice(self: *const Transitions) []const Transition {
|
||||
return self.items[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
/// Turns a stream of boot keyboard reports into make/break transitions by diffing
|
||||
/// each report against the last.
|
||||
pub const KeyboardDecoder = struct {
|
||||
previous: KeyboardReport = .{},
|
||||
|
||||
pub fn feed(self: *KeyboardDecoder, current: KeyboardReport) Transitions {
|
||||
var out = Transitions{};
|
||||
|
||||
// Rollover: 0x01 (ErrorRollOver) means more keys are held than the report
|
||||
// can carry, so the key array is invalid. Emit nothing and keep the prior
|
||||
// state (so the eventual releases still resolve against real keys).
|
||||
for (current.keys) |key| {
|
||||
if (key == 0x01) return out;
|
||||
}
|
||||
|
||||
// Modifiers: one make/break per changed bit; modifier usages are 224..231.
|
||||
const changed = current.modifiers ^ self.previous.modifiers;
|
||||
var bit: u3 = 0;
|
||||
while (true) : (bit += 1) {
|
||||
const mask = @as(u8, 1) << bit;
|
||||
if (changed & mask != 0) {
|
||||
out.add(.{
|
||||
.kind = if (current.modifiers & mask != 0) .pressed else .released,
|
||||
.usage = 224 + @as(u8, bit),
|
||||
});
|
||||
}
|
||||
if (bit == 7) break;
|
||||
}
|
||||
|
||||
// Keys made: present now, absent before.
|
||||
for (current.keys) |key| {
|
||||
if (key != 0 and !contains(&self.previous.keys, key)) out.add(.{ .kind = .pressed, .usage = key });
|
||||
}
|
||||
// Keys broken: present before, absent now.
|
||||
for (self.previous.keys) |key| {
|
||||
if (key != 0 and !contains(¤t.keys, key)) out.add(.{ .kind = .released, .usage = key });
|
||||
}
|
||||
|
||||
self.previous = current;
|
||||
return out;
|
||||
}
|
||||
};
|
||||
|
||||
fn contains(keys: *const [6]u8, value: u8) bool {
|
||||
for (keys) |key| {
|
||||
if (key == value) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// --- mouse ------------------------------------------------------------------
|
||||
|
||||
/// A decoded boot mouse report: the button bitmap and relative motion. The wheel
|
||||
/// byte is present only on 4-byte reports (QEMU's usb-mouse sends one).
|
||||
pub const MouseReport = struct {
|
||||
buttons: u8 = 0,
|
||||
dx: i8 = 0,
|
||||
dy: i8 = 0,
|
||||
wheel: i8 = 0,
|
||||
has_wheel: bool = false,
|
||||
};
|
||||
|
||||
pub const mouse_button_left: u8 = 1 << 0;
|
||||
pub const mouse_button_right: u8 = 1 << 1;
|
||||
pub const mouse_button_middle: u8 = 1 << 2;
|
||||
|
||||
/// Parse a 3- or 4-byte boot mouse report. Note HID reports Y in screen
|
||||
/// convention (positive = down), so — unlike PS/2 — `dy` is NOT negated.
|
||||
pub fn parseMouse(bytes: []const u8) ?MouseReport {
|
||||
if (bytes.len < 3) return null;
|
||||
return .{
|
||||
.buttons = bytes[0],
|
||||
.dx = @bitCast(bytes[1]),
|
||||
.dy = @bitCast(bytes[2]),
|
||||
.wheel = if (bytes.len >= 4) @bitCast(bytes[3]) else 0,
|
||||
.has_wheel = bytes.len >= 4,
|
||||
};
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
test "keyboard diff produces make and break transitions" {
|
||||
var decoder = KeyboardDecoder{};
|
||||
|
||||
// Press 'a' (usage 4).
|
||||
var t = decoder.feed(.{ .keys = .{ 4, 0, 0, 0, 0, 0 } });
|
||||
try std.testing.expectEqual(@as(usize, 1), t.count);
|
||||
try std.testing.expectEqual(TransitionKind.pressed, t.items[0].kind);
|
||||
try std.testing.expectEqual(@as(u8, 4), t.items[0].usage);
|
||||
|
||||
// Hold 'a', press 'b' (usage 5): only 'b' is new.
|
||||
t = decoder.feed(.{ .keys = .{ 4, 5, 0, 0, 0, 0 } });
|
||||
try std.testing.expectEqual(@as(usize, 1), t.count);
|
||||
try std.testing.expectEqual(@as(u8, 5), t.items[0].usage);
|
||||
|
||||
// Release everything: 'a' and 'b' both break.
|
||||
t = decoder.feed(.{ .keys = .{ 0, 0, 0, 0, 0, 0 } });
|
||||
try std.testing.expectEqual(@as(usize, 2), t.count);
|
||||
try std.testing.expectEqual(TransitionKind.released, t.items[0].kind);
|
||||
|
||||
// Press Left Shift (modifier bit 1 -> usage 225).
|
||||
t = decoder.feed(.{ .modifiers = modifier_left_shift });
|
||||
try std.testing.expectEqual(@as(usize, 1), t.count);
|
||||
try std.testing.expectEqual(@as(u8, 225), t.items[0].usage);
|
||||
try std.testing.expectEqual(TransitionKind.pressed, t.items[0].kind);
|
||||
}
|
||||
|
||||
test "rollover report is ignored but state is preserved" {
|
||||
var decoder = KeyboardDecoder{};
|
||||
_ = decoder.feed(.{ .keys = .{ 4, 0, 0, 0, 0, 0 } }); // press 'a'
|
||||
|
||||
const rollover = decoder.feed(.{ .keys = .{ 0x01, 0x01, 0x01, 0x01, 0x01, 0x01 } });
|
||||
try std.testing.expectEqual(@as(usize, 0), rollover.count);
|
||||
|
||||
// 'a' is still considered down, so releasing all keys now breaks it.
|
||||
const release = decoder.feed(.{ .keys = .{ 0, 0, 0, 0, 0, 0 } });
|
||||
try std.testing.expectEqual(@as(usize, 1), release.count);
|
||||
try std.testing.expectEqual(@as(u8, 4), release.items[0].usage);
|
||||
try std.testing.expectEqual(TransitionKind.released, release.items[0].kind);
|
||||
}
|
||||
|
||||
test "mouse report parses motion without inverting Y" {
|
||||
const three = parseMouse(&.{ mouse_button_left, 5, 0xFB }).?; // dy = -5
|
||||
try std.testing.expectEqual(mouse_button_left, three.buttons);
|
||||
try std.testing.expectEqual(@as(i8, 5), three.dx);
|
||||
try std.testing.expectEqual(@as(i8, -5), three.dy);
|
||||
try std.testing.expect(!three.has_wheel);
|
||||
|
||||
const four = parseMouse(&.{ 0, 0, 0, 0xFF }).?; // wheel = -1
|
||||
try std.testing.expect(four.has_wheel);
|
||||
try std.testing.expectEqual(@as(i8, -1), four.wheel);
|
||||
|
||||
try std.testing.expect(parseMouse(&.{ 0, 0 }) == null); // too short
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
//! USB HID boot keyboard driver.
|
||||
//!
|
||||
//! Spawned by the device manager when the xHCI bus driver reports a HID / boot /
|
||||
//! keyboard interface (class 3, subclass 1, protocol 1); its assigned device id
|
||||
//! arrives as argv[1] and an optional layout name ("us", "gb", ...) as argv[2].
|
||||
//! It owns no hardware: it opens its device through the USB transfer protocol
|
||||
//! (`runtime.usb`), asks the device for the boot protocol, subscribes to its
|
||||
//! interrupt-IN endpoint, and turns each 8-byte boot report into input-protocol
|
||||
//! events, published to the input service — the USB analogue of ps2-bus/keyboard.
|
||||
//!
|
||||
//! interrupt report -> hid-report diff -> key_down / key_up
|
||||
//! -> xkeyboard-config -> character -> key_press
|
||||
//!
|
||||
//! Because a USB keyboard's usages ARE the input protocol's keycodes (both are
|
||||
//! HID keyboard page 0x07), the decode is nearly 1:1 — no scancode translation.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const xkb = @import("xkeyboard-config");
|
||||
const hid = @import("hid-report.zig");
|
||||
const ipc = runtime.ipc;
|
||||
const process = runtime.process;
|
||||
const input_protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The modifier state a character lookup needs — derived from the report's
|
||||
// modifier byte, plus the driver-tracked caps-lock toggle.
|
||||
const ModifierSnapshot = struct {
|
||||
shift: bool,
|
||||
control: bool,
|
||||
right_alt: bool,
|
||||
caps_lock: bool,
|
||||
};
|
||||
|
||||
/// The character a key produces under `modifiers`, or 0 for none — the layout
|
||||
/// lookup for printable keys, with ASCII control characters for the keys every
|
||||
/// consumer expects (Enter, Tab, Backspace, Escape), exactly as ps2-bus/keyboard.
|
||||
fn characterFor(layout: *const xkb.Layout, usage: u8, modifiers: ModifierSnapshot) u32 {
|
||||
const mapping = xkb.map(layout, usage, .{
|
||||
.shift = modifiers.shift,
|
||||
.caps_lock = modifiers.caps_lock,
|
||||
.level3 = modifiers.right_alt,
|
||||
.control = modifiers.control,
|
||||
});
|
||||
if (mapping.character) |character| return character;
|
||||
return switch (@as(input_protocol.Keycode, @enumFromInt(usage))) {
|
||||
.enter, .keypad_enter => '\n',
|
||||
.tab => '\t',
|
||||
.backspace => 0x08,
|
||||
.escape => 0x1B,
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
fn modifierWord(modifiers: u8) u32 {
|
||||
var word: u32 = 0;
|
||||
if (modifiers & (hid.modifier_left_shift | hid.modifier_right_shift) != 0) word |= input_protocol.modifier_shift;
|
||||
if (modifiers & (hid.modifier_left_control | hid.modifier_right_control) != 0) word |= input_protocol.modifier_control;
|
||||
if (modifiers & (hid.modifier_left_alt | hid.modifier_right_alt) != 0) word |= input_protocol.modifier_alt;
|
||||
return word;
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-hid/keyboard: malformed device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
const layout = xkb.byName(init.arguments.get(2) orelse "us") orelse xkb.us;
|
||||
|
||||
// Hello the manager first (meet the spawn deadline), then open the device.
|
||||
if (!runtime.usb.helloManager(device_id)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: hello to device manager failed\n");
|
||||
return;
|
||||
}
|
||||
var device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-hid/keyboard: could not open device {d}\n", .{device_id});
|
||||
return;
|
||||
};
|
||||
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: no interrupt-IN endpoint\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// Ask for the boot protocol and an indefinite idle (report only on change).
|
||||
_ = device.controlOut(@bitCast(usb_abi.setProtocol(@enumFromInt(device.interface_number), .boot)));
|
||||
_ = device.controlOut(@bitCast(usb_abi.setIdle(@enumFromInt(device.interface_number), 0, 0)));
|
||||
|
||||
if (!device.subscribeInterrupt(endpoint.address, endpoint.max_packet_size)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: interrupt subscribe failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
var source = runtime.input.connectSource() orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/keyboard: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(device.endpoint);
|
||||
writeLine("/system/drivers/usb-hid/keyboard: ok (device {d}, interface {d}, layout {s})\n", .{ device_id, device.interface_number, layout.name });
|
||||
|
||||
var decoder = hid.KeyboardDecoder{};
|
||||
var caps_lock = false;
|
||||
var receive: [64]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(device.endpoint, &.{}, &receive, null);
|
||||
if (!got.isNotification()) continue;
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) return;
|
||||
continue;
|
||||
}
|
||||
if (!got.isMessage() or got.len < @sizeOf(runtime.usb.InterruptReport)) continue;
|
||||
|
||||
const message = std.mem.bytesToValue(runtime.usb.InterruptReport, receive[0..@sizeOf(runtime.usb.InterruptReport)]);
|
||||
if (message.length < @sizeOf(hid.KeyboardReport)) continue;
|
||||
const report = std.mem.bytesToValue(hid.KeyboardReport, message.data[0..@sizeOf(hid.KeyboardReport)]);
|
||||
const transitions = decoder.feed(report);
|
||||
|
||||
// Caps Lock toggles on its own key-down (a stateful lock, not a modifier).
|
||||
for (transitions.slice()) |transition| {
|
||||
if (transition.kind == .pressed and @as(input_protocol.Keycode, @enumFromInt(transition.usage)) == .caps_lock) caps_lock = !caps_lock;
|
||||
}
|
||||
|
||||
const modifiers = ModifierSnapshot{
|
||||
.shift = report.modifiers & (hid.modifier_left_shift | hid.modifier_right_shift) != 0,
|
||||
.control = report.modifiers & (hid.modifier_left_control | hid.modifier_right_control) != 0,
|
||||
.right_alt = report.modifiers & hid.modifier_right_alt != 0,
|
||||
.caps_lock = caps_lock,
|
||||
};
|
||||
const modifier_word = modifierWord(report.modifiers);
|
||||
|
||||
for (transitions.slice()) |transition| {
|
||||
switch (transition.kind) {
|
||||
.pressed => {
|
||||
_ = source.publishKeyboardEvent(.{
|
||||
.kind = @intFromEnum(input_protocol.EventKind.key_down),
|
||||
.keycode = transition.usage,
|
||||
.character = 0,
|
||||
.modifiers = modifier_word,
|
||||
});
|
||||
const character = characterFor(layout, transition.usage, modifiers);
|
||||
if (character != 0) {
|
||||
_ = source.publishKeyboardEvent(.{
|
||||
.kind = @intFromEnum(input_protocol.EventKind.key_press),
|
||||
.keycode = transition.usage,
|
||||
.character = character,
|
||||
.modifiers = modifier_word,
|
||||
});
|
||||
}
|
||||
},
|
||||
.released => {
|
||||
_ = source.publishKeyboardEvent(.{
|
||||
.kind = @intFromEnum(input_protocol.EventKind.key_up),
|
||||
.keycode = transition.usage,
|
||||
.character = 0,
|
||||
.modifiers = modifier_word,
|
||||
});
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
//! USB HID boot mouse driver.
|
||||
//!
|
||||
//! Spawned by the device manager when the xHCI bus driver reports a HID / boot /
|
||||
//! mouse interface (class 3, subclass 1, protocol 2); its assigned device id
|
||||
//! arrives as argv[1]. Like the keyboard driver it owns no hardware: it opens its
|
||||
//! device through the USB transfer protocol (`runtime.usb`), asks for the boot
|
||||
//! protocol, subscribes to its interrupt-IN endpoint, and turns each 3- or 4-byte
|
||||
//! boot report into input-protocol mouse events published to the input service.
|
||||
//!
|
||||
//! Unlike PS/2, HID reports Y in screen convention (positive = down), so motion
|
||||
//! is passed straight through (the decode in hid-report.zig does not negate it).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const hid = @import("hid-report.zig");
|
||||
const ipc = runtime.ipc;
|
||||
const process = runtime.process;
|
||||
const input_protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The current pressed-button bitmask in input-protocol terms.
|
||||
fn buttonMask(buttons: u8) u32 {
|
||||
var mask: u32 = 0;
|
||||
if (buttons & hid.mouse_button_left != 0) mask |= input_protocol.mouse_button_left;
|
||||
if (buttons & hid.mouse_button_right != 0) mask |= input_protocol.mouse_button_right;
|
||||
if (buttons & hid.mouse_button_middle != 0) mask |= input_protocol.mouse_button_middle;
|
||||
return mask;
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/mouse: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-hid/mouse: malformed device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
|
||||
if (!runtime.usb.helloManager(device_id)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/mouse: hello to device manager failed\n");
|
||||
return;
|
||||
}
|
||||
var device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-hid/mouse: could not open device {d}\n", .{device_id});
|
||||
return;
|
||||
};
|
||||
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/mouse: no interrupt-IN endpoint\n");
|
||||
return;
|
||||
};
|
||||
|
||||
_ = device.controlOut(@bitCast(usb_abi.setProtocol(@enumFromInt(device.interface_number), .boot)));
|
||||
|
||||
if (!device.subscribeInterrupt(endpoint.address, endpoint.max_packet_size)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/mouse: interrupt subscribe failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
var source = runtime.input.connectSource() orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-hid/mouse: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(device.endpoint);
|
||||
writeLine("/system/drivers/usb-hid/mouse: ok (device {d}, interface {d})\n", .{ device_id, device.interface_number });
|
||||
|
||||
var previous_buttons: u8 = 0;
|
||||
var receive: [64]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(device.endpoint, &.{}, &receive, null);
|
||||
if (!got.isNotification()) continue;
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) return;
|
||||
continue;
|
||||
}
|
||||
if (!got.isMessage() or got.len < @sizeOf(runtime.usb.InterruptReport)) continue;
|
||||
|
||||
const message = std.mem.bytesToValue(runtime.usb.InterruptReport, receive[0..@sizeOf(runtime.usb.InterruptReport)]);
|
||||
const length = @min(message.length, message.data.len);
|
||||
const report = hid.parseMouse(message.data[0..length]) orelse continue;
|
||||
const mask = buttonMask(report.buttons);
|
||||
|
||||
// Button transitions: one event per changed button bit.
|
||||
const changed = report.buttons ^ previous_buttons;
|
||||
inline for (.{
|
||||
.{ hid.mouse_button_left, input_protocol.mouse_button_left },
|
||||
.{ hid.mouse_button_right, input_protocol.mouse_button_right },
|
||||
.{ hid.mouse_button_middle, input_protocol.mouse_button_middle },
|
||||
}) |pair| {
|
||||
if (changed & pair[0] != 0) {
|
||||
_ = source.publishMouseEvent(.{
|
||||
.kind = @intFromEnum(if (report.buttons & pair[0] != 0) input_protocol.MouseEventKind.button_down else input_protocol.MouseEventKind.button_up),
|
||||
.button = pair[1],
|
||||
.dx = 0,
|
||||
.dy = 0,
|
||||
.scroll_x = 0,
|
||||
.scroll_y = 0,
|
||||
.buttons = mask,
|
||||
});
|
||||
}
|
||||
}
|
||||
previous_buttons = report.buttons;
|
||||
|
||||
// Relative motion (dy straight through — HID Y is already screen convention).
|
||||
if (report.dx != 0 or report.dy != 0) {
|
||||
_ = source.publishMouseEvent(.{
|
||||
.kind = @intFromEnum(input_protocol.MouseEventKind.motion),
|
||||
.button = 0,
|
||||
.dx = report.dx,
|
||||
.dy = report.dy,
|
||||
.scroll_x = 0,
|
||||
.scroll_y = 0,
|
||||
.buttons = mask,
|
||||
});
|
||||
}
|
||||
|
||||
// Wheel (4-byte reports only): positive = scroll up.
|
||||
if (report.has_wheel and report.wheel != 0) {
|
||||
_ = source.publishMouseEvent(.{
|
||||
.kind = @intFromEnum(input_protocol.MouseEventKind.scroll),
|
||||
.button = 0,
|
||||
.dx = 0,
|
||||
.dy = 0,
|
||||
.scroll_x = 0,
|
||||
.scroll_y = report.wheel,
|
||||
.buttons = mask,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
//! USB Mass Storage Bulk-Only Transport (BOT) wire structures — the Command and
|
||||
//! Command Status Wrappers that bracket every command (USB MSC BOT §5). Pure data
|
||||
//! definitions, host-testable in isolation. The command inside the CBW is a SCSI
|
||||
//! CDB (see scsi.zig); the transport here just carries it and reports status.
|
||||
//!
|
||||
//! One command is three bulk transfers: CBW out, an optional data stage, CSW in.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// "USBC" — the signature at the head of every Command Block Wrapper.
|
||||
pub const cbw_signature: u32 = 0x43425355;
|
||||
/// "USBS" — the signature at the head of every Command Status Wrapper.
|
||||
pub const csw_signature: u32 = 0x53425355;
|
||||
|
||||
/// CBW `flags`: set for a device-to-host (IN) data stage, clear for OUT.
|
||||
pub const flag_data_in: u8 = 0x80;
|
||||
|
||||
/// The 31-byte Command Block Wrapper, sent on the bulk-OUT endpoint.
|
||||
pub const CommandBlockWrapper = extern struct {
|
||||
signature: u32 align(1) = cbw_signature,
|
||||
tag: u32 align(1),
|
||||
data_transfer_length: u32 align(1),
|
||||
flags: u8,
|
||||
lun: u8,
|
||||
cdb_length: u8,
|
||||
cdb: [16]u8 = [_]u8{0} ** 16,
|
||||
};
|
||||
|
||||
/// A device's answer to a command (the CSW `status` byte).
|
||||
pub const CommandStatus = enum(u8) {
|
||||
passed = 0,
|
||||
failed = 1,
|
||||
phase_error = 2,
|
||||
_,
|
||||
};
|
||||
|
||||
/// The 13-byte Command Status Wrapper, read from the bulk-IN endpoint.
|
||||
pub const CommandStatusWrapper = extern struct {
|
||||
signature: u32 align(1) = csw_signature,
|
||||
tag: u32 align(1),
|
||||
data_residue: u32 align(1),
|
||||
status: u8,
|
||||
};
|
||||
|
||||
comptime {
|
||||
std.debug.assert(@sizeOf(CommandBlockWrapper) == 31);
|
||||
std.debug.assert(@sizeOf(CommandStatusWrapper) == 13);
|
||||
}
|
||||
|
||||
test "wrapper sizes and signatures match the specification" {
|
||||
const cbw = CommandBlockWrapper{
|
||||
.tag = 0x11223344,
|
||||
.data_transfer_length = 512,
|
||||
.flags = flag_data_in,
|
||||
.lun = 0,
|
||||
.cdb_length = 10,
|
||||
};
|
||||
const bytes = std.mem.asBytes(&cbw);
|
||||
try std.testing.expectEqual(@as(usize, 31), bytes.len);
|
||||
// "USBC" little-endian.
|
||||
try std.testing.expectEqualSlices(u8, "USBC", bytes[0..4]);
|
||||
try std.testing.expectEqual(flag_data_in, bytes[12]);
|
||||
|
||||
const csw = std.mem.bytesToValue(CommandStatusWrapper, &[_]u8{
|
||||
0x55, 0x53, 0x42, 0x53, // "USBS"
|
||||
0x44, 0x33, 0x22, 0x11, // tag
|
||||
0x00, 0x00, 0x00, 0x00, // residue
|
||||
0x00, // passed
|
||||
});
|
||||
try std.testing.expectEqual(csw_signature, csw.signature);
|
||||
try std.testing.expectEqual(@as(u32, 0x11223344), csw.tag);
|
||||
try std.testing.expectEqual(@as(u8, @intFromEnum(CommandStatus.passed)), csw.status);
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
//! The SCSI command descriptor blocks a transparent-SCSI (subclass 0x06) mass
|
||||
//! storage device understands, and the parsers for what they return. Pure data —
|
||||
//! host-testable. These CDBs go inside a Bulk-Only-Transport CBW (see
|
||||
//! bulk-only-transport.zig).
|
||||
//!
|
||||
//! Every multi-byte SCSI field is **big-endian** — the opposite of the USB wire
|
||||
//! ABI — so the LBA and transfer-length encodings are the load-bearing detail.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// SCSI operation codes.
|
||||
const op_test_unit_ready: u8 = 0x00;
|
||||
const op_request_sense: u8 = 0x03;
|
||||
const op_inquiry: u8 = 0x12;
|
||||
const op_read_capacity_10: u8 = 0x25;
|
||||
const op_read_10: u8 = 0x28;
|
||||
const op_write_10: u8 = 0x2A;
|
||||
const op_synchronize_cache_10: u8 = 0x35;
|
||||
|
||||
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
||||
/// and product strings).
|
||||
pub fn inquiry(allocation_length: u8) [6]u8 {
|
||||
return .{ op_inquiry, 0, 0, 0, allocation_length, 0 };
|
||||
}
|
||||
|
||||
/// TEST UNIT READY: no data; success (CSW passed) means the unit is ready.
|
||||
pub fn testUnitReady() [6]u8 {
|
||||
return .{ op_test_unit_ready, 0, 0, 0, 0, 0 };
|
||||
}
|
||||
|
||||
/// REQUEST SENSE: 18 bytes of sense data (sense key + ASC/ASCQ) explaining the
|
||||
/// previous failure.
|
||||
pub fn requestSense(allocation_length: u8) [6]u8 {
|
||||
return .{ op_request_sense, 0, 0, 0, allocation_length, 0 };
|
||||
}
|
||||
|
||||
/// READ CAPACITY(10): 8 bytes back — the last LBA and the block size, both u32
|
||||
/// big-endian. Block count is last_lba + 1.
|
||||
pub fn readCapacity10() [10]u8 {
|
||||
return .{ op_read_capacity_10, 0, 0, 0, 0, 0, 0, 0, 0, 0 };
|
||||
}
|
||||
|
||||
/// READ(10): read `blocks` logical blocks starting at `lba` into the data stage.
|
||||
pub fn read10(lba: u32, blocks: u16) [10]u8 {
|
||||
var cdb = [_]u8{0} ** 10;
|
||||
cdb[0] = op_read_10;
|
||||
std.mem.writeInt(u32, cdb[2..6], lba, .big);
|
||||
std.mem.writeInt(u16, cdb[7..9], blocks, .big);
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// WRITE(10): write `blocks` logical blocks starting at `lba` from the data stage.
|
||||
pub fn write10(lba: u32, blocks: u16) [10]u8 {
|
||||
var cdb = [_]u8{0} ** 10;
|
||||
cdb[0] = op_write_10;
|
||||
std.mem.writeInt(u32, cdb[2..6], lba, .big);
|
||||
std.mem.writeInt(u16, cdb[7..9], blocks, .big);
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE(10): commit the device's write cache to stable media. LBA 0
|
||||
/// and block count 0 mean "the whole medium". No data stage. Without this a write
|
||||
/// can sit in the USB flash controller's cache and be lost if power is cut right
|
||||
/// after — which is exactly what a shutdown-time log flush hits on real hardware.
|
||||
pub fn synchronizeCache10() [10]u8 {
|
||||
var cdb = [_]u8{0} ** 10;
|
||||
cdb[0] = op_synchronize_cache_10;
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// Decode an 8-byte READ CAPACITY(10) reply.
|
||||
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
||||
return .{
|
||||
.last_lba = std.mem.readInt(u32, bytes[0..4], .big),
|
||||
.block_size = std.mem.readInt(u32, bytes[4..8], .big),
|
||||
};
|
||||
}
|
||||
|
||||
test "read/write CDBs encode the LBA and length big-endian" {
|
||||
const read = read10(0x01020304, 8);
|
||||
try std.testing.expectEqualSlices(u8, &.{ 0x28, 0x00, 0x01, 0x02, 0x03, 0x04, 0x00, 0x00, 0x08, 0x00 }, &read);
|
||||
|
||||
const write = write10(0xAABBCCDD, 1);
|
||||
try std.testing.expectEqualSlices(u8, &.{ 0x2A, 0x00, 0xAA, 0xBB, 0xCC, 0xDD, 0x00, 0x00, 0x01, 0x00 }, &write);
|
||||
|
||||
try std.testing.expectEqual(@as(u8, 0x25), readCapacity10()[0]);
|
||||
try std.testing.expectEqual(@as(u8, 0x12), inquiry(36)[0]);
|
||||
try std.testing.expectEqual(@as(u8, 36), inquiry(36)[4]);
|
||||
try std.testing.expectEqual(@as(u8, 0x00), testUnitReady()[0]);
|
||||
}
|
||||
|
||||
test "read capacity parses last LBA and block size" {
|
||||
// last_lba = 0x0003FFFF (262144 blocks), block_size = 512.
|
||||
const capacity = parseCapacity(.{ 0x00, 0x03, 0xFF, 0xFF, 0x00, 0x00, 0x02, 0x00 });
|
||||
try std.testing.expectEqual(@as(u32, 0x0003FFFF), capacity.last_lba);
|
||||
try std.testing.expectEqual(@as(u32, 512), capacity.block_size);
|
||||
}
|
||||
@@ -0,0 +1,187 @@
|
||||
//! USB mass-storage class driver (Bulk-Only Transport + transparent SCSI).
|
||||
//!
|
||||
//! Spawned by the device manager when the xHCI bus driver reports a mass-storage
|
||||
//! / SCSI / bulk-only interface (class 8, subclass 6, protocol 0x50); its device
|
||||
//! id arrives as argv[1]. It owns no hardware: it opens its device through the
|
||||
//! USB transfer protocol (`runtime.usb`), then drives it with the BOT command
|
||||
//! cycle — CBW out, an optional data stage, CSW in — carrying SCSI commands
|
||||
//! (READ CAPACITY, READ(10), WRITE(10)). Upward it is a block device: it serves
|
||||
//! the block protocol under `.block`, the storage a FAT filesystem sits on.
|
||||
//!
|
||||
//! Block data never crosses IPC: read/write name a caller-owned DMA buffer by
|
||||
//! physical address, which the data stage DMAs straight to/from.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const scsi = @import("scsi.zig");
|
||||
const bot = @import("bulk-only-transport.zig");
|
||||
const block_protocol = @import("block-protocol");
|
||||
const dma = runtime.dma;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
var device_id: u64 = 0;
|
||||
var device: runtime.usb.Device = undefined;
|
||||
var bulk_in: runtime.usb.Endpoint = undefined;
|
||||
var bulk_out: runtime.usb.Endpoint = undefined;
|
||||
|
||||
// DMA buffers for the transport: the 31-byte CBW, the 13-byte CSW, and a page
|
||||
// for the small command data (INQUIRY / READ CAPACITY / the self-check sector).
|
||||
var command_wrapper: dma.Region = undefined;
|
||||
var status_wrapper: dma.Region = undefined;
|
||||
var command_data: dma.Region = undefined;
|
||||
|
||||
var next_tag: u32 = 1;
|
||||
var block_size: u32 = 512;
|
||||
var block_count: u64 = 0;
|
||||
|
||||
/// One Bulk-Only-Transport command: send the CBW, run the data stage (to/from
|
||||
/// `data_physical`), read and validate the CSW. Returns true on a passed status.
|
||||
fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length: u32) bool {
|
||||
const tag = next_tag;
|
||||
next_tag +%= 1;
|
||||
|
||||
const wrapper: *bot.CommandBlockWrapper = @ptrFromInt(command_wrapper.virtual);
|
||||
wrapper.* = .{
|
||||
.tag = tag,
|
||||
.data_transfer_length = data_length,
|
||||
.flags = if (direction_in) bot.flag_data_in else 0,
|
||||
.lun = 0,
|
||||
.cdb_length = @intCast(cdb.len),
|
||||
};
|
||||
@memcpy(wrapper.cdb[0..cdb.len], cdb);
|
||||
|
||||
if (device.bulk(bulk_out.address, command_wrapper.physical, @sizeOf(bot.CommandBlockWrapper)) == null) return false;
|
||||
if (data_length > 0) {
|
||||
const endpoint = if (direction_in) bulk_in.address else bulk_out.address;
|
||||
if (device.bulk(endpoint, data_physical, data_length) == null) return false;
|
||||
}
|
||||
if (device.bulk(bulk_in.address, status_wrapper.physical, @sizeOf(bot.CommandStatusWrapper)) == null) return false;
|
||||
|
||||
const status: *const bot.CommandStatusWrapper = @ptrFromInt(status_wrapper.virtual);
|
||||
if (status.signature != bot.csw_signature or status.tag != tag) return false;
|
||||
return status.status == @intFromEnum(bot.CommandStatus.passed);
|
||||
}
|
||||
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!runtime.usb.helloManager(device_id)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: hello to device manager failed\n");
|
||||
return false;
|
||||
}
|
||||
device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-storage: could not open device {d}\n", .{device_id});
|
||||
return false;
|
||||
};
|
||||
bulk_in = device.findEndpoint(runtime.usb.transfer_type_bulk, true) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: no bulk-IN endpoint\n");
|
||||
return false;
|
||||
};
|
||||
bulk_out = device.findEndpoint(runtime.usb.transfer_type_bulk, false) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: no bulk-OUT endpoint\n");
|
||||
return false;
|
||||
};
|
||||
command_wrapper = dma.alloc(4096, dma.coherent) orelse return false;
|
||||
status_wrapper = dma.alloc(4096, dma.coherent) orelse return false;
|
||||
command_data = dma.alloc(4096, dma.coherent) orelse return false;
|
||||
|
||||
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
||||
// with REQUEST SENSE), identify it, and read its capacity.
|
||||
var tries: u32 = 0;
|
||||
while (tries < 10) : (tries += 1) {
|
||||
const ready = scsi.testUnitReady();
|
||||
if (transact(&ready, false, 0, 0)) break;
|
||||
const sense = scsi.requestSense(18);
|
||||
_ = transact(&sense, true, command_data.physical, 18);
|
||||
runtime.system.sleep(50);
|
||||
}
|
||||
const inquiry = scsi.inquiry(36);
|
||||
_ = transact(&inquiry, true, command_data.physical, 36);
|
||||
|
||||
const capacity_command = scsi.readCapacity10();
|
||||
if (!transact(&capacity_command, true, command_data.physical, 8)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: READ CAPACITY failed\n");
|
||||
return false;
|
||||
}
|
||||
var capacity_bytes: [8]u8 = undefined;
|
||||
const capacity_source: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||
@memcpy(&capacity_bytes, capacity_source[0..8]);
|
||||
const capacity = scsi.parseCapacity(capacity_bytes);
|
||||
block_size = capacity.block_size;
|
||||
block_count = @as(u64, capacity.last_lba) + 1;
|
||||
writeLine("/system/drivers/usb-storage: ready ({d} blocks x {d} bytes)\n", .{ block_count, block_size });
|
||||
|
||||
// Self-check: read block 0 and log its trailing signature (0x55AA for a boot
|
||||
// sector) — proof READ(10) works end to end over the bulk path.
|
||||
const read0 = scsi.read10(0, 1);
|
||||
if (block_size <= 4096 and transact(&read0, true, command_data.physical, block_size)) {
|
||||
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||
writeLine("/system/drivers/usb-storage: block 0 signature 0x{x:0>2}{x:0>2}\n", .{ sector[510], sector[511] });
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Serve the block protocol: geometry, and whole-block read/write to/from the
|
||||
/// caller's DMA buffer (named by physical address).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < block_protocol.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.read) => {
|
||||
const count: u16 = @intCast(request.count);
|
||||
const cdb = scsi.read10(@intCast(request.lba), count);
|
||||
const ok = transact(&cdb, true, request.physical, request.count * block_size);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.write) => {
|
||||
const count: u16 = @intCast(request.count);
|
||||
const cdb = scsi.write10(@intCast(request.lba), count);
|
||||
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.flush) => {
|
||||
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||
// cuts power. A device without a volatile cache reports success anyway.
|
||||
const cdb = scsi.synchronizeCache10();
|
||||
const ok = transact(&cdb, false, 0, 0);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
fn writeReply(reply: []u8, value: block_protocol.Reply) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
return bytes.len;
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-storage: malformed device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(block_protocol.message_maximum, .{
|
||||
.service = .block,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,159 @@
|
||||
//! The USB transfer protocol: what a USB class driver (a keyboard, mouse, or
|
||||
//! mass-storage driver) says to the xHCI bus driver over its well-known
|
||||
//! `.usb_bus` endpoint to drive its device. The class driver owns no hardware —
|
||||
//! it reaches its device entirely through these messages, the way a PS/2 keyboard
|
||||
//! driver reaches the 8042 through the ps2-bus. Extern-struct messages tagged by
|
||||
//! `Operation`, the vfs-protocol / device-manager-protocol pattern.
|
||||
//!
|
||||
//! The shape:
|
||||
//! - **open** (a capability-passing `ipc.callCap`): the class driver hands over
|
||||
//! its own endpoint (for asynchronous interrupt reports) and its assigned
|
||||
//! device id, and receives a `device_token` plus its interface's endpoints.
|
||||
//! - **control / bulk** (synchronous `ipc.call`): one transfer, answered when
|
||||
//! it completes. Control data travels inline (descriptors, HID/MSC class
|
||||
//! requests are all small); bulk data travels by **physical address** — the
|
||||
//! class driver's own `dma_alloc`'d buffer — so a 512-byte sector never has
|
||||
//! to cross the 256-byte IPC boundary.
|
||||
//! - **interrupt_subscribe** (synchronous): arm periodic IN polling of an
|
||||
//! interrupt endpoint; each report the device produces is then pushed to the
|
||||
//! class driver's endpoint as an asynchronous `InterruptReport` (`ipc.send`),
|
||||
//! exactly how the input service delivers events.
|
||||
//!
|
||||
//! Single controller assumption: one `.usb_bus` singleton serves QEMU's one xHCI.
|
||||
//! A multi-controller machine would need a per-controller endpoint (the device
|
||||
//! manager handing each class driver the right one); noted, not built.
|
||||
|
||||
/// Fits one synchronous IPC message (kernel MESSAGE_MAXIMUM).
|
||||
pub const message_maximum: usize = 256;
|
||||
|
||||
/// The largest inline control-transfer payload. Sized so a whole message
|
||||
/// (header + data) stays under `message_maximum`: descriptors and HID/MSC class
|
||||
/// requests are all far smaller.
|
||||
pub const max_inline_data: usize = 200;
|
||||
|
||||
/// The largest interrupt report pushed asynchronously. Sized so `InterruptReport`
|
||||
/// fits an `ipc_send` payload slot (POST_MAXIMUM = 64): boot keyboard reports are
|
||||
/// 8 bytes, boot mouse reports 3–4.
|
||||
pub const max_report_data: usize = 48;
|
||||
|
||||
/// Endpoints per interface reported back in an open reply (a boot HID interface
|
||||
/// has one interrupt endpoint, a mass-storage interface two bulk endpoints).
|
||||
pub const max_reported_endpoints: usize = 4;
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
open = 0,
|
||||
control = 1,
|
||||
interrupt_subscribe = 2,
|
||||
bulk = 3,
|
||||
};
|
||||
|
||||
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||
/// the bus driver already parsed during enumeration.
|
||||
pub const Endpoint = extern struct {
|
||||
/// EndpointDescriptor address: direction in bit 7, number in bits 3:0.
|
||||
address: u8,
|
||||
/// 0 control, 1 isochronous, 2 bulk, 3 interrupt.
|
||||
transfer_type: u8,
|
||||
max_packet_size: u16,
|
||||
interval: u8,
|
||||
reserved: [3]u8 = .{ 0, 0, 0 },
|
||||
};
|
||||
|
||||
/// open: the class driver's receive endpoint rides as the call's capability, and
|
||||
/// `device_id` is the interface's assigned id (its argv[1]).
|
||||
pub const OpenRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.open),
|
||||
reserved: u32 = 0,
|
||||
device_id: u64,
|
||||
};
|
||||
|
||||
/// The answer to open: a token scoping every later request to this device, the
|
||||
/// interface's class triple (a sanity check), and its endpoints.
|
||||
pub const OpenReply = extern struct {
|
||||
status: i32,
|
||||
endpoint_count: u32,
|
||||
device_token: u64,
|
||||
interface_class: u8,
|
||||
interface_subclass: u8,
|
||||
interface_protocol: u8,
|
||||
interface_number: u8,
|
||||
reserved2: u32 = 0,
|
||||
endpoints: [max_reported_endpoints]Endpoint = [_]Endpoint{.{ .address = 0, .transfer_type = 0, .max_packet_size = 0, .interval = 0 }} ** max_reported_endpoints,
|
||||
};
|
||||
|
||||
/// control: one EP0 control transfer. `setup` is a bit-cast `usb_abi.Request`.
|
||||
/// For an OUT transfer `data[0..data_length]` is sent; for an IN transfer the
|
||||
/// reply carries up to `data_length` bytes back.
|
||||
pub const ControlRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.control),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
setup: [8]u8,
|
||||
direction_in: u8, // 1 = device-to-host (IN), 0 = host-to-device (OUT)
|
||||
reserved2: u8 = 0,
|
||||
data_length: u16,
|
||||
reserved3: u32 = 0,
|
||||
data: [max_inline_data]u8 = [_]u8{0} ** max_inline_data,
|
||||
};
|
||||
|
||||
pub const ControlReply = extern struct {
|
||||
status: i32, // 0 success, negative on failure/stall
|
||||
actual_length: u32,
|
||||
data: [max_inline_data]u8 = [_]u8{0} ** max_inline_data,
|
||||
};
|
||||
|
||||
/// interrupt_subscribe: begin periodic IN polling of an interrupt endpoint. Each
|
||||
/// report the device returns is pushed to the caller's endpoint (handed over at
|
||||
/// open) as an asynchronous `InterruptReport`.
|
||||
pub const InterruptSubscribeRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.interrupt_subscribe),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
endpoint_address: u8,
|
||||
reserved2: u8 = 0,
|
||||
max_length: u16, // bytes to request per poll (the endpoint's max packet size)
|
||||
};
|
||||
|
||||
pub const InterruptSubscribeReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// bulk: one bulk IN or OUT transfer. `physical_address` is the class driver's own
|
||||
/// `dma_alloc`'d buffer — the controller DMAs straight to/from it, so the bulk
|
||||
/// data never crosses IPC. `endpoint_address`'s bit 7 selects IN vs OUT.
|
||||
pub const BulkRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.bulk),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
physical_address: u64,
|
||||
length: u32,
|
||||
endpoint_address: u8,
|
||||
reserved2: u8 = 0,
|
||||
reserved3: u16 = 0,
|
||||
};
|
||||
|
||||
pub const BulkReply = extern struct {
|
||||
status: i32,
|
||||
actual_length: u32,
|
||||
};
|
||||
|
||||
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||
pub const InterruptReport = extern struct {
|
||||
device_token: u64,
|
||||
endpoint_address: u8,
|
||||
length: u8,
|
||||
reserved: u16 = 0,
|
||||
data: [max_report_data]u8 = [_]u8{0} ** max_report_data,
|
||||
};
|
||||
|
||||
comptime {
|
||||
const std = @import("std");
|
||||
// Every synchronous message must fit one IPC message; the async report must
|
||||
// fit an ipc_send payload slot.
|
||||
std.debug.assert(@sizeOf(ControlRequest) <= message_maximum);
|
||||
std.debug.assert(@sizeOf(ControlReply) <= message_maximum);
|
||||
std.debug.assert(@sizeOf(OpenReply) <= message_maximum);
|
||||
std.debug.assert(@sizeOf(InterruptReport) <= 64);
|
||||
}
|
||||
@@ -17,6 +17,52 @@ const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const usb_ids = @import("usb-ids");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const transfer = @import("usb-transfer-protocol");
|
||||
const library = @import("usb-xhci-library.zig");
|
||||
|
||||
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||
var controller: ?library.Controller = null;
|
||||
|
||||
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||
/// requests, signals, and the interrupt-poll timer all arrive.
|
||||
var service_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
||||
/// re-armed each tick. Frequent enough for responsive input.
|
||||
const poll_interval_ms: u64 = 8;
|
||||
|
||||
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||
const Open = struct {
|
||||
used: bool = false,
|
||||
device_token: u64 = 0,
|
||||
report_endpoint: usize = 0,
|
||||
};
|
||||
var opens = [_]Open{.{}} ** 16;
|
||||
|
||||
fn recordOpen(device_token: u64, report_endpoint: usize) void {
|
||||
for (&opens) |*open| {
|
||||
if (open.used and open.device_token == device_token) {
|
||||
open.report_endpoint = report_endpoint;
|
||||
return;
|
||||
}
|
||||
}
|
||||
for (&opens) |*open| {
|
||||
if (!open.used) {
|
||||
open.* = .{ .used = true, .device_token = device_token, .report_endpoint = report_endpoint };
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn reportEndpointFor(device_token: u64) ?usize {
|
||||
for (&opens) |*open| {
|
||||
if (open.used and open.device_token == device_token) return open.report_endpoint;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so
|
||||
/// concurrent instances (one per controller) can never interleave mid-line.
|
||||
@@ -31,22 +77,22 @@ var controller_id: u64 = protocol.no_device;
|
||||
/// manager. Any failure returns false: the process exits cleanly, which the
|
||||
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
service_endpoint = endpoint;
|
||||
if (!device.claim(controller_id)) {
|
||||
writeLine("usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||
writeLine("/system/drivers/usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the controller's resources.
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("usb-xhci-bus: out of memory\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: out of memory\n");
|
||||
return false;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
writeLine("usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||
writeLine("/system/drivers/usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
@@ -59,19 +105,39 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
break resource;
|
||||
}
|
||||
} else {
|
||||
writeLine("usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
writeLine("usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||
writeLine("/system/drivers/usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||
controller_id,
|
||||
register_window.start,
|
||||
register_window.len,
|
||||
});
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
_ = runtime.system.write("usb-xhci-bus: mmio_map failed\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: mmio_map failed\n");
|
||||
return false;
|
||||
};
|
||||
|
||||
// Bring the controller up: reset it, stand up the command and event rings,
|
||||
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||
controller = library.Controller.init(register_base) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller running ({d} slots, {d}-byte contexts)\n", .{
|
||||
controller.?.max_slots,
|
||||
controller.?.context_size,
|
||||
});
|
||||
// The proof of life: a No-Op command round-trips the command ring, the event
|
||||
// ring, the doorbell, and the cycle-bit bookkeeping. If this completes, the
|
||||
// engine is sound; transfers build on exactly this machinery.
|
||||
if (controller.?.noOpCommand()) {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: command ring running (no-op ok)\n");
|
||||
} else {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no-op command did not complete\n");
|
||||
return false;
|
||||
}
|
||||
|
||||
// The handshake: role, protocol version, assignment — inside the manager's
|
||||
// deadline (the lookup retries cover the manager still registering).
|
||||
var manager: ?runtime.ipc.Handle = null;
|
||||
@@ -81,91 +147,274 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
if (manager == null) runtime.system.sleep(20);
|
||||
}
|
||||
const h = manager orelse {
|
||||
_ = runtime.system.write("usb-xhci-bus: no device manager to hello\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no device manager to hello\n");
|
||||
return false;
|
||||
};
|
||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||
_ = runtime.system.write("usb-xhci-bus: hello call failed\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello call failed\n");
|
||||
return false;
|
||||
};
|
||||
if (n < protocol.reply_size or std.mem.bytesToValue(protocol.HelloReply, reply[0..protocol.reply_size]).status != 0) {
|
||||
_ = runtime.system.write("usb-xhci-bus: hello refused\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello refused\n");
|
||||
return false;
|
||||
}
|
||||
_ = runtime.system.write("usb-xhci-bus: hello acknowledged\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: hello acknowledged\n");
|
||||
|
||||
scanPorts(h);
|
||||
|
||||
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
||||
// re-armed on each tick in onNotification; class drivers subscribe later.
|
||||
_ = runtime.system.timerOnce(service_endpoint, poll_interval_ms);
|
||||
return true;
|
||||
}
|
||||
|
||||
var register_base: usize = 0;
|
||||
|
||||
/// One 32-bit volatile register read at `offset` from the mapped window.
|
||||
fn readRegister(offset: usize) u32 {
|
||||
const register: *volatile u32 = @ptrFromInt(register_base + offset);
|
||||
return register.*;
|
||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||
/// decoded to human names — the boot-log breadcrumb for what actually enumerated on
|
||||
/// a port, the USB analog of the pci-bus class-code line. A controller may redefine
|
||||
/// these through its Supported Protocol capability, but the defaults cover every
|
||||
/// speed QEMU and real hardware report at this (pre-descriptor) stage.
|
||||
fn speedName(speed: u32) []const u8 {
|
||||
return switch (speed) {
|
||||
1 => "Full-speed (USB 2.0, 12 Mb/s)",
|
||||
2 => "Low-speed (USB 2.0, 1.5 Mb/s)",
|
||||
3 => "High-speed (USB 2.0, 480 Mb/s)",
|
||||
4 => "SuperSpeed (USB 3.0, 5 Gb/s)",
|
||||
5 => "SuperSpeedPlus (USB 3.1, 10 Gb/s)",
|
||||
else => "unknown speed",
|
||||
};
|
||||
}
|
||||
|
||||
/// The root-hub port scan: read the capability registers for the port count
|
||||
/// and the operational-register offset, then one PORTSC per port. The connect
|
||||
/// bit (CCS) and the speed field reflect hardware state directly — no
|
||||
/// controller reset or run needed to *see* the devices; driving them needs the
|
||||
/// rings (the USB track).
|
||||
/// The root-hub scan and enumeration: for each connected port, bring the device
|
||||
/// up (reset → enable slot → address), read its descriptors, and register +
|
||||
/// report one child per interface — carrying the interface's (class, subclass,
|
||||
/// protocol) triple as identity, which is what the device manager matches a
|
||||
/// class driver against.
|
||||
fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||
// Capability registers: CAPLENGTH is byte 0 of the first dword; HCSPARAMS1
|
||||
// carries MaxPorts in bits 31:24.
|
||||
const capability_length = readRegister(0) & 0xFF;
|
||||
const structural = readRegister(0x04);
|
||||
const maximum_ports: u32 = structural >> 24;
|
||||
writeLine("usb-xhci-bus: {d} root-hub ports\n", .{maximum_ports});
|
||||
const engine = if (controller) |*c| c else {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller not initialised\n");
|
||||
return;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: {d} root-hub ports\n", .{engine.max_ports});
|
||||
|
||||
// PORTSC registers: operational base + 0x400 + 0x10 per port (1-based).
|
||||
var port: u32 = 1;
|
||||
var connected: u32 = 0;
|
||||
while (port <= maximum_ports) : (port += 1) {
|
||||
const port_status = readRegister(capability_length + 0x400 + 0x10 * (port - 1));
|
||||
while (port <= engine.max_ports) : (port += 1) {
|
||||
const port_status = engine.portStatus(port);
|
||||
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
||||
connected += 1;
|
||||
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
||||
writeLine("usb-xhci-bus: port {d} connected (speed class {d})\n", .{ port, speed });
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} connected — {s} (speed class {d})\n", .{ port, speedName(speed), speed });
|
||||
|
||||
const report = protocol.ChildAdded{
|
||||
.parent = controller_id,
|
||||
.bus_address = port,
|
||||
.identity = speed,
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("usb-xhci-bus: child report for port {d} failed\n", .{port});
|
||||
const usb_device = engine.setupDevice(port, speed) orelse {
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} device setup failed\n", .{port});
|
||||
continue;
|
||||
};
|
||||
if (!engine.enumerate(usb_device)) {
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} enumeration failed\n", .{port});
|
||||
continue;
|
||||
}
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} device vendor 0x{x:0>4} product 0x{x:0>4}, {d} interface(s)\n", .{
|
||||
port,
|
||||
usb_device.device_descriptor.vendor_id,
|
||||
usb_device.device_descriptor.product_id,
|
||||
usb_device.interface_count,
|
||||
});
|
||||
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||
// Record the id each interface was registered as, so a class driver
|
||||
// opening the interface (by that id) resolves to it.
|
||||
if (reportInterface(manager, port, interface.*)) |registered| {
|
||||
interface.registered_device_id = registered;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (connected == 0) _ = runtime.system.write("usb-xhci-bus: no devices connected\n");
|
||||
if (connected == 0) _ = runtime.system.write("/system/drivers/usb-xhci-bus: no devices connected\n");
|
||||
}
|
||||
|
||||
/// No bus protocol to serve yet — transfer requests arrive with the USB track.
|
||||
/// Register one interface as a resource-less child of the controller and report
|
||||
/// it to the device manager. The identity is the packed USB class triple, so the
|
||||
/// manager can match a class driver (HID keyboard, mouse, mass storage); the
|
||||
/// registered device id becomes that driver's argv[1] assignment. Returns the
|
||||
/// registered device id, or null if registration or the report failed.
|
||||
fn reportInterface(manager: runtime.ipc.Handle, port: u32, interface: library.InterfaceInfo) ?u64 {
|
||||
const identity = usb_ids.packTriple(interface.class, interface.subclass, interface.protocol);
|
||||
|
||||
// A USB device is reached through its controller, not by MMIO, so the child
|
||||
// carries no resources; register() allows that. Its bus-local identity — the
|
||||
// (port, interface) address, written as a short "P<port>I<interface>" tag in
|
||||
// the hid field — makes each interface a distinct kernel node (the register
|
||||
// dedup keys on class/pci_class/hid/resources, all otherwise identical here)
|
||||
// and keeps re-registration idempotent across a bus restart: the same port
|
||||
// and interface always map back to the same device id.
|
||||
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||
descriptor.class = @intFromEnum(device.DeviceClass.usb_device);
|
||||
descriptor.pci_class = device.no_pci_class;
|
||||
descriptor.resource_count = 0;
|
||||
var hid_buffer: [8]u8 = undefined;
|
||||
const hid_text = std.fmt.bufPrint(&hid_buffer, "P{d}I{d}", .{ port, interface.number }) catch "";
|
||||
descriptor.hid_len = hid_text.len;
|
||||
@memcpy(descriptor.hid[0..hid_text.len], hid_text);
|
||||
const registered = device.register(controller_id, &descriptor) orelse {
|
||||
writeLine("/system/drivers/usb-xhci-bus: register refused for port {d} interface {d}\n", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
|
||||
const report = protocol.ChildAdded{
|
||||
.parent = controller_id,
|
||||
.bus_address = (@as(u64, port) << 8) | interface.number,
|
||||
.identity = identity,
|
||||
.device_id = registered,
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: child report for port {d} interface {d} failed\n", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} interface {d} class {d}/{d}/{d} registered as device {d}\n", .{
|
||||
port,
|
||||
interface.number,
|
||||
interface.class,
|
||||
interface.subclass,
|
||||
interface.protocol,
|
||||
registered,
|
||||
});
|
||||
return registered;
|
||||
}
|
||||
|
||||
/// Serve the USB transfer protocol: a class driver opens its device, then issues
|
||||
/// control / interrupt-subscribe / bulk requests against it.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0;
|
||||
if (message.len < 4) return 0;
|
||||
const operation = std.mem.readInt(u32, message[0..4], .little);
|
||||
return switch (operation) {
|
||||
@intFromEnum(transfer.Operation.open) => handleOpen(message, reply, capability),
|
||||
@intFromEnum(transfer.Operation.control) => handleControl(message, reply),
|
||||
@intFromEnum(transfer.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||
@intFromEnum(transfer.Operation.bulk) => handleBulk(message, reply),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
fn writeReply(reply: []u8, value: anytype) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
return bytes.len;
|
||||
}
|
||||
|
||||
/// open: resolve the assigned device id to an interface, remember the caller's
|
||||
/// endpoint (for interrupt reports), and answer with a device token + the
|
||||
/// interface's endpoints so the class driver need not re-read the config.
|
||||
fn handleOpen(message: []const u8, reply: []u8, capability: ?runtime.ipc.Handle) usize {
|
||||
if (message.len < @sizeOf(transfer.OpenRequest)) return writeReply(reply, transfer.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||
const request = std.mem.bytesToValue(transfer.OpenRequest, message[0..@sizeOf(transfer.OpenRequest)]);
|
||||
const engine = if (controller) |*c| c else return writeReply(reply, transfer.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||
const found = engine.findInterface(request.device_id) orelse return writeReply(reply, transfer.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||
|
||||
if (capability) |endpoint| recordOpen(request.device_id, endpoint);
|
||||
|
||||
var open_reply = transfer.OpenReply{
|
||||
.status = 0,
|
||||
.endpoint_count = found.interface.endpoint_count,
|
||||
.device_token = request.device_id,
|
||||
.interface_class = found.interface.class,
|
||||
.interface_subclass = found.interface.subclass,
|
||||
.interface_protocol = found.interface.protocol,
|
||||
.interface_number = found.interface.number,
|
||||
};
|
||||
const count = @min(found.interface.endpoint_count, transfer.max_reported_endpoints);
|
||||
for (found.interface.endpoints[0..count], 0..) |endpoint, index| {
|
||||
open_reply.endpoints[index] = .{
|
||||
.address = endpoint.address,
|
||||
.transfer_type = endpoint.transfer_type,
|
||||
.max_packet_size = endpoint.max_packet_size,
|
||||
.interval = endpoint.interval,
|
||||
};
|
||||
}
|
||||
return writeReply(reply, open_reply);
|
||||
}
|
||||
|
||||
/// control: one EP0 control transfer, small data inline both ways.
|
||||
fn handleControl(message: []const u8, reply: []u8) usize {
|
||||
if (message.len < @sizeOf(transfer.ControlRequest)) return writeReply(reply, transfer.ControlReply{ .status = -1, .actual_length = 0 });
|
||||
const request = std.mem.bytesToValue(transfer.ControlRequest, message[0..@sizeOf(transfer.ControlRequest)]);
|
||||
const engine = if (controller) |*c| c else return writeReply(reply, transfer.ControlReply{ .status = -1, .actual_length = 0 });
|
||||
const found = engine.findInterface(request.device_token) orelse return writeReply(reply, transfer.ControlReply{ .status = -1, .actual_length = 0 });
|
||||
|
||||
const setup = std.mem.bytesToValue(usb_abi.Request, &request.setup);
|
||||
const direction_in = request.direction_in != 0;
|
||||
const data_length = @min(request.data_length, transfer.max_inline_data);
|
||||
var data: [transfer.max_inline_data]u8 = undefined;
|
||||
if (!direction_in) @memcpy(data[0..data_length], request.data[0..data_length]);
|
||||
|
||||
const ok = engine.controlTransfer(found.device, setup, data[0..data_length], direction_in);
|
||||
var control_reply = transfer.ControlReply{ .status = if (ok) 0 else -1, .actual_length = if (ok) data_length else 0 };
|
||||
if (ok and direction_in) @memcpy(control_reply.data[0..data_length], data[0..data_length]);
|
||||
return writeReply(reply, control_reply);
|
||||
}
|
||||
|
||||
/// interrupt_subscribe: arm periodic IN polling; reports flow back asynchronously.
|
||||
fn handleSubscribe(message: []const u8, reply: []u8) usize {
|
||||
if (message.len < @sizeOf(transfer.InterruptSubscribeRequest)) return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||
const request = std.mem.bytesToValue(transfer.InterruptSubscribeRequest, message[0..@sizeOf(transfer.InterruptSubscribeRequest)]);
|
||||
const engine = if (controller) |*c| c else return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||
const found = engine.findInterface(request.device_token) orelse return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||
const endpoint = library.Controller.endpointForAddress(found.interface, request.endpoint_address) orelse return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||
const report_endpoint = reportEndpointFor(request.device_token) orelse return writeReply(reply, transfer.InterruptSubscribeReply{ .status = -1 });
|
||||
const ok = engine.subscribeInterrupt(found.device, endpoint, request.device_token, report_endpoint);
|
||||
return writeReply(reply, transfer.InterruptSubscribeReply{ .status = if (ok) 0 else -1 });
|
||||
}
|
||||
|
||||
/// bulk: one bulk transfer to/from the class driver's own DMA buffer (by physical
|
||||
/// address), so sector-sized data never crosses IPC.
|
||||
fn handleBulk(message: []const u8, reply: []u8) usize {
|
||||
if (message.len < @sizeOf(transfer.BulkRequest)) return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||
const request = std.mem.bytesToValue(transfer.BulkRequest, message[0..@sizeOf(transfer.BulkRequest)]);
|
||||
const engine = if (controller) |*c| c else return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||
const found = engine.findInterface(request.device_token) orelse return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||
const endpoint = library.Controller.endpointForAddress(found.interface, request.endpoint_address) orelse return writeReply(reply, transfer.BulkReply{ .status = -1, .actual_length = 0 });
|
||||
const transferred = engine.bulkTransfer(found.device, endpoint, request.physical_address, request.length);
|
||||
return writeReply(reply, transfer.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||
}
|
||||
|
||||
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
||||
/// each to the class driver that subscribed, then re-arm the timer.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & runtime.ipc.notify_timer_bit == 0) return;
|
||||
if (controller) |*engine| {
|
||||
engine.pump();
|
||||
while (engine.takeReport()) |report| {
|
||||
var message = transfer.InterruptReport{
|
||||
.device_token = report.device_token,
|
||||
.endpoint_address = report.endpoint_address,
|
||||
.length = @intCast(@min(report.length, transfer.max_report_data)),
|
||||
};
|
||||
const n = @min(report.length, transfer.max_report_data);
|
||||
@memcpy(message.data[0..n], report.data[0..n]);
|
||||
_ = runtime.ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||
}
|
||||
}
|
||||
_ = runtime.system.timerOnce(service_endpoint, poll_interval_ms);
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = runtime.system.write("usb-xhci-bus: missing controller device id (argv[1])\n");
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: missing controller device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||
writeLine("/system/drivers/usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
runtime.service.run(transfer.message_maximum, .{
|
||||
.service = .usb_bus,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -80,6 +80,35 @@ var timer_hz: u32 = 0;
|
||||
var tsc_hz: u64 = 0;
|
||||
var tsc_base: u64 = 0;
|
||||
|
||||
/// Whether the TSC is architecturally **invariant** — a constant rate regardless of
|
||||
/// P/C-state transitions, and thus valid as a clocksource (CPUID leaf 0x80000007,
|
||||
/// EDX bit 8). AMD and modern Intel set it; the bare qemu64 model does not. Measured
|
||||
/// frequency alone is not enough: a non-invariant TSC speeds up and slows down with
|
||||
/// the core clock, so reading it as wall time would drift.
|
||||
var tsc_invariant: bool = false;
|
||||
/// Cleared if the cross-core warp check (checkWarpSource) ever sees the TSC read
|
||||
/// lower on one core than the max another core has already published — i.e. the
|
||||
/// per-core TSCs are not synchronized, and a task migrating cores could see time go
|
||||
/// backward. Starts true (assume synchronized until proven otherwise).
|
||||
var tsc_synced: bool = true;
|
||||
/// The worst backward skew the warp check observed, in TSC cycles (0 = none).
|
||||
var tsc_warp_cycles: u64 = 0;
|
||||
|
||||
/// The monotonic clock's source. The TSC when it is invariant *and* synchronized —
|
||||
/// the fast `rdtsc` path taken on real Intel/AMD and modern VMs. Otherwise the HPET
|
||||
/// main counter: a single fixed-rate counter, immune to both per-core skew and
|
||||
/// frequency scaling, so it stays accurate on a bare VM or a warped machine.
|
||||
const ClockSource = enum { tsc, hpet };
|
||||
var clock_source: ClockSource = .tsc;
|
||||
|
||||
/// HPET standby clocksource, set up in calibrate() whenever an HPET exists (whether
|
||||
/// or not calibration itself measured against it): its frequency, the counter value
|
||||
/// chosen as the zero point, and its width mask. Only a 64-bit HPET is used as a
|
||||
/// clocksource — a 32-bit one wraps too fast to be monotonic without accumulation.
|
||||
var hpet_clock_hz: u64 = 0;
|
||||
var hpet_clock_base: u64 = 0;
|
||||
var hpet_clock_mask: u64 = ~@as(u64, 0);
|
||||
|
||||
/// Read the 64-bit Time Stamp Counter.
|
||||
fn rdtsc() u64 {
|
||||
var low: u32 = undefined;
|
||||
@@ -220,6 +249,31 @@ pub fn calibrate() void {
|
||||
}
|
||||
|
||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||
|
||||
// Decide whether the TSC is trustworthy as a clocksource. Frequency (measured
|
||||
// above, possibly against the HPET/PIT) is necessary but not sufficient: the TSC
|
||||
// must also be *invariant* (CPUID 0x80000007 EDX[8]). AMD and modern Intel set
|
||||
// this; the bare qemu64 model does not.
|
||||
tsc_invariant = tscIsInvariant();
|
||||
|
||||
// Bring up the HPET as a standby clocksource whenever one exists — even on the
|
||||
// CPUID-0x15 path where calibration never touched it — so a non-invariant TSC
|
||||
// (here) or an unsynchronized one (checkWarpSource, during SMP bring-up) can fall
|
||||
// back to a source that is immune to both. hpetHz() maps + enables the counter
|
||||
// and is idempotent if calibration already used it.
|
||||
if (configuration_hpet_base != 0) {
|
||||
if (hpetHz()) |hz| {
|
||||
hpet_clock_mask = hpetMask();
|
||||
if (hpet_clock_mask == ~@as(u64, 0)) { // only a 64-bit HPET is monotonic enough
|
||||
hpet_clock_hz = hz;
|
||||
hpet_clock_base = readHpet();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Select the source: the fast TSC when invariant, else the HPET if we have one.
|
||||
// (checkWarpSource may still demote TSC -> HPET later if the cores' TSCs skew.)
|
||||
if (!tsc_invariant and hpet_clock_hz != 0) clock_source = .hpet;
|
||||
}
|
||||
|
||||
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||
@@ -275,7 +329,7 @@ fn calibratePit() void {
|
||||
// --- reference clocks ------------------------------------------------------
|
||||
|
||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU, and on AMD).
|
||||
fn cpuidTscHz() ?u64 {
|
||||
if (cpuid(0).eax < 0x15) return null;
|
||||
const r = cpuid(0x15);
|
||||
@@ -283,6 +337,15 @@ fn cpuidTscHz() ?u64 {
|
||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||
}
|
||||
|
||||
/// Whether the CPU advertises an **invariant** TSC (CPUID leaf 0x80000007, EDX
|
||||
/// bit 8) — the architectural guarantee, on both Intel and AMD, that the TSC ticks
|
||||
/// at a constant rate across P/C-states and never stops. Requires the extended-leaf
|
||||
/// range to reach 0x80000007 first.
|
||||
fn tscIsInvariant() bool {
|
||||
if (cpuid(0x80000000).eax < 0x80000007) return false;
|
||||
return (cpuid(0x80000007).edx & (1 << 8)) != 0;
|
||||
}
|
||||
|
||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||
|
||||
fn cpuid(leaf: u32) CpuidRegs {
|
||||
@@ -310,11 +373,20 @@ fn hpetWrite64(off: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||
}
|
||||
|
||||
/// Whether the HPET has been mapped into the physmap yet, so `configuration_hpet_base`
|
||||
/// already holds the virtual address. `hpetHz` is called more than once (calibration
|
||||
/// may use the HPET, and the standby-clocksource setup asks for it again), and mapping
|
||||
/// an already-mapped base a second time would double-offset it into an overflow.
|
||||
var hpet_mapped: bool = false;
|
||||
|
||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||
/// address, so the register accessors reach it without the identity map.
|
||||
/// address, so the register accessors reach it without the identity map. Idempotent.
|
||||
fn hpetHz() ?u64 {
|
||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||
if (!hpet_mapped) {
|
||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||
hpet_mapped = true;
|
||||
}
|
||||
const caps = hpetRead64(0x00);
|
||||
const period_fs = caps >> 32; // femtoseconds per tick
|
||||
if (period_fs == 0) return null;
|
||||
@@ -364,24 +436,169 @@ pub fn tscHz() u64 {
|
||||
return tsc_hz;
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
||||
// the scheduler uses for sleep deadlines.
|
||||
// Monotonic high-resolution clock. A function per resolution, each scaling the
|
||||
// counter delta directly at its unit (the 128-bit intermediate avoids overflow
|
||||
// across a long uptime). nanos() resolves to a few ns on the TSC; millis() is what
|
||||
// the scheduler uses for sleep deadlines. The source is the TSC when it is invariant
|
||||
// and synchronized, else the HPET counter (see clock_source) — the branch is one
|
||||
// global load and the TSC path is unchanged from before.
|
||||
|
||||
/// The selected source's counter delta since its zero point.
|
||||
fn clockCount() u64 {
|
||||
return switch (clock_source) {
|
||||
.tsc => rdtsc() -% tsc_base,
|
||||
// A 64-bit HPET (the only kind we select) never wraps in any realistic
|
||||
// uptime, so the wrapping subtraction is exact.
|
||||
.hpet => readHpet() -% hpet_clock_base,
|
||||
};
|
||||
}
|
||||
|
||||
/// The selected source's frequency (0 if the clock is unavailable/uncalibrated).
|
||||
fn clockHertz() u64 {
|
||||
return switch (clock_source) {
|
||||
.tsc => tsc_hz,
|
||||
.hpet => hpet_clock_hz,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn nanos() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
||||
const hz = clockHertz();
|
||||
if (hz == 0) return 0;
|
||||
return @intCast(@as(u128, clockCount()) * 1_000_000_000 / hz);
|
||||
}
|
||||
|
||||
pub fn micros() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
||||
const hz = clockHertz();
|
||||
if (hz == 0) return 0;
|
||||
return @intCast(@as(u128, clockCount()) * 1_000_000 / hz);
|
||||
}
|
||||
|
||||
pub fn millis() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
||||
const hz = clockHertz();
|
||||
if (hz == 0) return 0;
|
||||
return @intCast(@as(u128, clockCount()) * 1_000 / hz);
|
||||
}
|
||||
|
||||
/// Whether the CPU advertises an invariant TSC (CPUID 0x80000007 EDX[8]).
|
||||
pub fn tscInvariant() bool {
|
||||
return tsc_invariant;
|
||||
}
|
||||
|
||||
/// Test hook: force the TSC clocksource on, as if the CPU had advertised an invariant
|
||||
/// TSC. QEMU's TCG accelerator (the only one for an x86 guest on an Apple-Silicon
|
||||
/// host) does not expose the invariant-TSC bit — its emulated TSC isn't invariant — so
|
||||
/// the tsc-sync test can't reach the real-Intel/AMD/KVM path through CPUID. This lets
|
||||
/// that test exercise the TSC clocksource and the cross-core warp check anyway. tsc_base
|
||||
/// is left as-is so the switch from the HPET is continuous.
|
||||
pub fn forceTscClocksourceForTest() void {
|
||||
tsc_invariant = true;
|
||||
clock_source = .tsc;
|
||||
}
|
||||
|
||||
/// How many per-AP warp checks actually ran (a rendezvous completed) — lets a test
|
||||
/// confirm the cross-core check executed rather than being skipped.
|
||||
pub fn warpChecksRun() u32 {
|
||||
return warp_checks;
|
||||
}
|
||||
|
||||
/// Whether the per-core TSCs are synchronized (no backward warp seen at bring-up).
|
||||
pub fn tscSynced() bool {
|
||||
return tsc_synced;
|
||||
}
|
||||
|
||||
/// The active monotonic clocksource, for the boot log and tests.
|
||||
pub fn clockSourceName() []const u8 {
|
||||
return switch (clock_source) {
|
||||
.tsc => "tsc",
|
||||
.hpet => "hpet",
|
||||
};
|
||||
}
|
||||
|
||||
// --- cross-core TSC synchronization ("warp") check -------------------------
|
||||
// Two cores hammer a shared "max seen" TSC value under a lock; if either reads a
|
||||
// value below that max, its TSC lags the other's, and time would run backward for a
|
||||
// task migrating between them (Linux calls this a warp). danos brings APs up one at a
|
||||
// time, so this runs pairwise: the BSP (source) against each AP (target) as it comes
|
||||
// online. It only matters — and only runs — while the TSC is the clocksource; on a
|
||||
// machine already on the HPET (a bare VM) the whole rendezvous is skipped.
|
||||
|
||||
var warp_lock: u32 = 0;
|
||||
var warp_last: u64 = 0;
|
||||
var warp_bsp_ready: u32 = 0;
|
||||
var warp_ap_ready: u32 = 0;
|
||||
var warp_stop: u32 = 0;
|
||||
var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test)
|
||||
|
||||
const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates
|
||||
const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot
|
||||
|
||||
fn warpTick() void {
|
||||
while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause");
|
||||
const t = rdtsc();
|
||||
if (t < warp_last) {
|
||||
const delta = warp_last - t;
|
||||
if (delta > tsc_warp_cycles) tsc_warp_cycles = delta;
|
||||
tsc_synced = false;
|
||||
} else {
|
||||
warp_last = t;
|
||||
}
|
||||
@atomicStore(u32, &warp_lock, 0, .release);
|
||||
}
|
||||
|
||||
/// Spin (bounded) until `flag` is nonzero; false on timeout.
|
||||
fn warpAwait(flag: *u32) bool {
|
||||
var spins: u64 = 0;
|
||||
while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) {
|
||||
if (spins >= warp_spin_limit) return false;
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// BSP side of the pairwise TSC warp check, run once per AP as it reports in. No-op
|
||||
/// unless the TSC is the active clocksource. If the AP's TSC proves to lag, demote
|
||||
/// the monotonic clock to the HPET without a discontinuity.
|
||||
pub fn checkWarpSource() void {
|
||||
if (clock_source != .tsc) return;
|
||||
warp_last = 0;
|
||||
@atomicStore(u32, &warp_stop, 0, .release);
|
||||
@atomicStore(u32, &warp_ap_ready, 0, .release);
|
||||
@atomicStore(u32, &warp_bsp_ready, 1, .release);
|
||||
if (!warpAwait(&warp_ap_ready)) { // AP never joined the rendezvous; skip, don't hang
|
||||
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||
return;
|
||||
}
|
||||
var i: u32 = 0;
|
||||
while (i < warp_rounds) : (i += 1) warpTick();
|
||||
@atomicStore(u32, &warp_stop, 1, .release);
|
||||
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||
warp_checks += 1;
|
||||
|
||||
if (!tsc_synced and hpet_clock_hz != 0) demoteToHpet();
|
||||
}
|
||||
|
||||
/// AP side: join the BSP's warp check, then return so the core can enter the
|
||||
/// scheduler. Bounded so a missing BSP can't strand the core.
|
||||
pub fn checkWarpTarget() void {
|
||||
if (clock_source != .tsc) return;
|
||||
if (!warpAwait(&warp_bsp_ready)) return;
|
||||
@atomicStore(u32, &warp_ap_ready, 1, .release);
|
||||
var spins: u64 = 0;
|
||||
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) {
|
||||
if (spins >= warp_spin_limit) return;
|
||||
warpTick();
|
||||
}
|
||||
}
|
||||
|
||||
/// Switch the clocksource from the TSC to the HPET without a discontinuity: choose
|
||||
/// the HPET zero point so it reads the same nanosecond value the TSC does right now,
|
||||
/// so time neither jumps nor runs backward across the switch. Called when the warp
|
||||
/// check proves the per-core TSCs unsynchronized.
|
||||
fn demoteToHpet() void {
|
||||
const now_ns = @as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz;
|
||||
const equivalent_ticks: u64 = @intCast(now_ns * hpet_clock_hz / 1_000_000_000);
|
||||
hpet_clock_base = readHpet() -% equivalent_ticks;
|
||||
clock_source = .hpet;
|
||||
}
|
||||
|
||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||
|
||||
@@ -106,6 +106,12 @@ pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Whether a working UART was detected (loopback probe). When false the serial
|
||||
/// sink is silently inert — a dead legacy COM1 costs nothing per byte.
|
||||
pub fn serialPresent() bool {
|
||||
return serial.present();
|
||||
}
|
||||
|
||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||
/// displays. The last-resort progress signal when there's no text output at all.
|
||||
@@ -469,6 +475,124 @@ pub fn clockHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
// --- real-time clock (CMOS) --------------------------------------------------
|
||||
//
|
||||
// The battery-backed CMOS clock, read once at boot and thereafter anchored to the
|
||||
// monotonic clock (see kernel/wall-clock.zig) — so this is never on a hot path and
|
||||
// needs no lock. Wall-clock *seconds* are mechanism the kernel owns (the hardware's
|
||||
// value), like the monotonic clock; calendars/timezones are policy layered on top.
|
||||
|
||||
fn cmosRead(register: u8) u8 {
|
||||
io.outb(0x70, register);
|
||||
return io.inb(0x71);
|
||||
}
|
||||
|
||||
const RtcFields = struct { second: u8, minute: u8, hour: u8, day: u8, month: u8, year: u8 };
|
||||
|
||||
fn rtcRaw() RtcFields {
|
||||
while (cmosRead(0x0A) & 0x80 != 0) {} // wait out any update in progress (status A bit 7)
|
||||
return .{
|
||||
.second = cmosRead(0x00),
|
||||
.minute = cmosRead(0x02),
|
||||
.hour = cmosRead(0x04),
|
||||
.day = cmosRead(0x07),
|
||||
.month = cmosRead(0x08),
|
||||
.year = cmosRead(0x09),
|
||||
};
|
||||
}
|
||||
|
||||
fn bcdToBinary(v: u8) u8 {
|
||||
return (v & 0x0F) + ((v >> 4) * 10);
|
||||
}
|
||||
|
||||
fn isLeapYear(y: u32) bool {
|
||||
return (y % 4 == 0 and y % 100 != 0) or (y % 400 == 0);
|
||||
}
|
||||
|
||||
/// Read the CMOS real-time clock and convert it to Unix epoch seconds (UTC).
|
||||
pub fn readRtcUnixSeconds() u64 {
|
||||
// Read until two consecutive reads agree, so we never latch a half-updated time.
|
||||
var a = rtcRaw();
|
||||
while (true) {
|
||||
const b = rtcRaw();
|
||||
if (a.second == b.second and a.minute == b.minute and a.hour == b.hour and
|
||||
a.day == b.day and a.month == b.month and a.year == b.year) break;
|
||||
a = b;
|
||||
}
|
||||
|
||||
const status_b = cmosRead(0x0B);
|
||||
const binary_mode = status_b & 0x04 != 0; // else BCD
|
||||
const hour_24 = status_b & 0x02 != 0; // else 12-hour with a PM bit
|
||||
|
||||
var second = a.second;
|
||||
var minute = a.minute;
|
||||
var hour_field = a.hour;
|
||||
var day = a.day;
|
||||
var month = a.month;
|
||||
var year = a.year;
|
||||
if (!binary_mode) {
|
||||
second = bcdToBinary(second);
|
||||
minute = bcdToBinary(minute);
|
||||
hour_field = bcdToBinary(hour_field & 0x7F) | (hour_field & 0x80); // preserve the PM bit
|
||||
day = bcdToBinary(day);
|
||||
month = bcdToBinary(month);
|
||||
year = bcdToBinary(year);
|
||||
}
|
||||
|
||||
var hour: u32 = hour_field & 0x7F;
|
||||
if (!hour_24) {
|
||||
const pm = hour_field & 0x80 != 0;
|
||||
hour %= 12; // 12 AM/PM -> 0
|
||||
if (pm) hour += 12;
|
||||
}
|
||||
|
||||
// The CMOS year is 0..99; QEMU and modern hardware mean 20xx (there is no
|
||||
// reliable century register on QEMU). Treat < 70 as 20xx, else 19xx.
|
||||
const full_year: u32 = if (year < 70) 2000 + @as(u32, year) else 1900 + @as(u32, year);
|
||||
|
||||
var days: u64 = 0;
|
||||
var y: u32 = 1970;
|
||||
while (y < full_year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||
const month_lengths = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||
var m: u8 = 1;
|
||||
while (m < month) : (m += 1) {
|
||||
days += month_lengths[m - 1];
|
||||
if (m == 2 and isLeapYear(full_year)) days += 1;
|
||||
}
|
||||
days += @as(u64, day) - 1;
|
||||
|
||||
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||
}
|
||||
|
||||
/// Whether the CPU guarantees an **invariant** TSC (CPUID 0x80000007 EDX[8] on
|
||||
/// x86; the analogous architectural guarantee elsewhere). When false the TSC is not
|
||||
/// used as the clocksource.
|
||||
pub fn clockInvariant() bool {
|
||||
return apic.tscInvariant();
|
||||
}
|
||||
|
||||
/// Whether the per-core clock counters are synchronized (no backward warp observed
|
||||
/// at SMP bring-up). When false the clock falls back off the TSC.
|
||||
pub fn clockSynchronized() bool {
|
||||
return apic.tscSynced();
|
||||
}
|
||||
|
||||
/// The active monotonic clocksource, for the boot log ("tsc" or "hpet" on x86).
|
||||
pub fn clockSourceName() []const u8 {
|
||||
return apic.clockSourceName();
|
||||
}
|
||||
|
||||
/// Test hook: force the TSC clocksource on, to exercise the TSC + warp-check path on
|
||||
/// a hypervisor that won't advertise an invariant TSC (see apic.forceTscClocksourceForTest).
|
||||
pub fn forceTscClocksourceForTest() void {
|
||||
apic.forceTscClocksourceForTest();
|
||||
}
|
||||
|
||||
/// How many per-AP TSC warp checks completed (for the tsc-sync test).
|
||||
pub fn warpChecksRun() u32 {
|
||||
return apic.warpChecksRun();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
|
||||
@@ -7,8 +7,13 @@
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
//! The physmap (the permanent window onto all physical RAM) is built with 2 MiB
|
||||
//! huge pages wherever the range is 2 MiB-aligned, falling back to 4 KiB for the
|
||||
//! unaligned edges. On a big machine that is the difference between ~16.7M page-
|
||||
//! table entries (128 MiB of tables) and ~32K — it makes both the build and the
|
||||
//! footprint scale sanely with RAM. Everything else (kernel segments, heap, user
|
||||
//! space, on-demand MMIO) stays 4 KiB: precise, and the table memory is
|
||||
//! negligible there.
|
||||
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
@@ -22,10 +27,22 @@ const writable: u64 = 1 << 1;
|
||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||
const pwt: u64 = 1 << 3; // page write-through
|
||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||
const page_size_bit: u64 = 1 << 7; // PS: this PDPT/PD entry is a 1 GiB/2 MiB leaf, not a pointer to the next table
|
||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// The PAT-index bit. In a 4 KiB PTE it is bit 7; in a huge leaf (2 MiB PDE / 1 GiB
|
||||
// PDPTE) bit 7 is PS, so the PAT bit moves to bit 12. With PCD=PWT=0 this selects
|
||||
// PAT entry 4, which `setupPat` programs to write-combining (see mapRangePhysmap).
|
||||
const pte_pat: u64 = 1 << 7;
|
||||
const huge_pat: u64 = 1 << 12;
|
||||
const ia32_pat: u32 = 0x277;
|
||||
|
||||
/// The physmap's page size for 2 MiB-aligned RAM: one PD leaf covers this instead
|
||||
/// of 512 PT entries. 4 KiB pages fill the unaligned edges (see mapRangePhysmap).
|
||||
const huge_page_size: u64 = 2 << 20; // 2 MiB
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
@@ -74,7 +91,15 @@ fn allocTable() u64 {
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & address_mask;
|
||||
if (entry.* & present != 0) {
|
||||
// A present-but-huge entry is a leaf, not a table: descending would read
|
||||
// its 2 MiB/1 GiB data frame as a page table and corrupt RAM. This only
|
||||
// fires on a bug — a 4 KiB map landing inside a physmap huge page — and a
|
||||
// loud panic beats silent corruption. (The physmap and the 4 KiB regions
|
||||
// live in disjoint PML4 slots, so it should never happen.)
|
||||
if (entry.* & page_size_bit != 0) @panic("paging: descend through a huge-page leaf");
|
||||
return entry.* & address_mask;
|
||||
}
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
@@ -96,15 +121,39 @@ fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Map one 2 MiB huge page `virtual` -> `physical` with `flags` — a leaf at the PD
|
||||
/// level (PS bit set), with no PT beneath it. Both addresses must be 2 MiB-aligned.
|
||||
/// One of these replaces 512 `mapPage`s (and the PT frame they'd need).
|
||||
fn mapHugePage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||
@panic("paging: new higher-half PML4 entry after init");
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
tableAt(pd)[(virtual >> 21) & 0x1FF] = (physical & address_mask) | flags | present | page_size_bit;
|
||||
}
|
||||
|
||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||
/// window onto physical memory once the low identity map goes away.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||
/// window onto physical memory once the low identity map goes away. The 2 MiB-
|
||||
/// aligned interior is mapped with huge pages; the unaligned head/tail with 4 KiB.
|
||||
/// `write_combining` selects the WC memory type (setupPat's PAT entry 4) via the
|
||||
/// PAT bit — bit 7 in a 4 KiB PTE, bit 12 in a huge leaf — for the framebuffer.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64, write_combining: bool) void {
|
||||
const pte_flags = if (write_combining) flags | pte_pat else flags;
|
||||
const huge_flags = if (write_combining) flags | huge_pat else flags;
|
||||
var address = physical_base & ~@as(u64, page_size - 1);
|
||||
const end = physical_base + len;
|
||||
while (address < end) : (address += page_size) {
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
||||
}
|
||||
// Head: 4 KiB pages up to the next 2 MiB boundary.
|
||||
while (address < end and address & (huge_page_size - 1) != 0) : (address += page_size)
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||
// Interior: 2 MiB huge pages while a whole one still fits.
|
||||
while (address + huge_page_size <= end) : (address += huge_page_size)
|
||||
mapHugePage(pml4, boot_handoff.physicalToVirtual(address), address, huge_flags);
|
||||
// Tail: 4 KiB pages for whatever is left.
|
||||
while (address < end) : (address += page_size)
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||
}
|
||||
|
||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||
@@ -118,11 +167,26 @@ fn enableNx() void {
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Program this core's PAT so entry 4 (selected by the PAT bit with PCD=PWT=0) is
|
||||
/// **write-combining**, leaving the other seven at their reset types. Nothing else
|
||||
/// in danos sets the PAT bit, so this changes no existing mapping — it only gives
|
||||
/// the framebuffer a write-combining type, which turns its full-screen clear from
|
||||
/// glacial (uncached writes to a GPU BAR, the real-hardware default via MTRRs) into
|
||||
/// a batched burst. Must run on **every** core (PAT is per-logical-processor) — the
|
||||
/// framebuffer mapping lives in the shared kernel half, so a core with the reset
|
||||
/// PAT would see it as write-back and alias. Called from `init` (BSP) and each AP.
|
||||
pub fn setupPat() void {
|
||||
// Reset PAT is PA0=WB PA1=WT PA2=UC- PA3=UC PA4=WB PA5=WT PA6=UC- PA7=UC; flip
|
||||
// PA4 from WB (0x06) to WC (0x01). Type codes: UC=0 WC=1 WT=4 WP=5 WB=6 UC-=7.
|
||||
io.wrmsr(ia32_pat, 0x0007_0401_0007_0406);
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||
alloc_frame = allocFrame;
|
||||
free_frame = freeFrame;
|
||||
enableNx();
|
||||
setupPat(); // BSP: PAT entry 4 = write-combining, for the framebuffer window
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||
@@ -130,13 +194,15 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
||||
// mapped on demand (mapMmio) or explicitly below.
|
||||
for (regions(boot_information.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute, false);
|
||||
}
|
||||
|
||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||
// the kernel touches directly), RW + NX.
|
||||
// the kernel touches directly), RW + NX. The framebuffer is **write-
|
||||
// combining** (see setupPat) so the console's full-screen clear is a burst,
|
||||
// not millions of uncached single-word writes.
|
||||
const fb = boot_information.framebuffer;
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute, true);
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||
@@ -338,8 +404,8 @@ fn freeSubtree(physical: u64, level: u32) void {
|
||||
}
|
||||
|
||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||
/// clear. Walks the 4-level tables, stopping at a 2 MiB huge-page leaf (the physmap
|
||||
/// uses them). Returns false if unmapped. Used for W^X checks in tests.
|
||||
pub fn isExecutable(virtual: u64) bool {
|
||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return false;
|
||||
@@ -347,6 +413,7 @@ pub fn isExecutable(virtual: u64) bool {
|
||||
if (pdpte & present == 0) return false;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return false;
|
||||
if (pde & page_size_bit != 0) return pde & no_execute == 0; // 2 MiB huge leaf
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return false;
|
||||
return pte & no_execute == 0;
|
||||
@@ -385,9 +452,9 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||
/// needs the frame behind a user vaddr to free it).
|
||||
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
@@ -395,6 +462,8 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
if (pdpte & present == 0) return null;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return null;
|
||||
if (pde & page_size_bit != 0) // 2 MiB huge leaf: frame base is bits 51:21
|
||||
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return null;
|
||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||
|
||||
@@ -17,6 +17,12 @@ const Access = enum { port, mmio };
|
||||
var access: Access = .port;
|
||||
var base: u64 = 0x3F8; // COM1
|
||||
|
||||
/// Whether `init`/`reconfigure` found a *working* UART at `base`. False on a
|
||||
/// legacy-free machine whose COM1 is decoded but dead: writing to it is then a
|
||||
/// no-op, so `write` never spins waiting for a transmit register that will never
|
||||
/// drain. Cleared until proven by the loopback probe.
|
||||
var uart_present: bool = false;
|
||||
|
||||
fn portOut(p: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[p]"
|
||||
:
|
||||
@@ -57,6 +63,34 @@ pub fn init() void {
|
||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||
setRegister(4, 0x0B); // RTS/DSR set
|
||||
uart_present = probe();
|
||||
}
|
||||
|
||||
/// Detect a *working* UART by internal loopback: route the transmitter back to
|
||||
/// the receiver (MCR bit 4), send a byte, and check it comes back. A port that is
|
||||
/// merely decoded but has nothing behind it (the common case on a legacy-free
|
||||
/// board that still answers I/O at 0x3F8) never echoes, so this returns false.
|
||||
///
|
||||
/// This matters for speed, not just correctness: a dead UART's line-status
|
||||
/// register reads back 0x00, so its transmit-holding-empty bit never sets, and
|
||||
/// `writeByte` would otherwise spin its full guard — tens of milliseconds — on
|
||||
/// *every* logged byte. On real hardware that alone can add ~a minute to boot.
|
||||
fn probe() bool {
|
||||
const saved_mcr = register(4);
|
||||
setRegister(4, 0x1E); // MCR: LOOP | OUT2 | OUT1 | RTS — internal loopback
|
||||
setRegister(0, 0xAE); // push a distinctive byte into the loopback path
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x01 == 0 and guard < 10_000) : (guard += 1) {} // await Data Ready
|
||||
const echo = register(0);
|
||||
setRegister(4, saved_mcr); // restore the modem-control lines
|
||||
return echo == 0xAE;
|
||||
}
|
||||
|
||||
/// Whether a working UART was detected (see `probe`). The log sink stays
|
||||
/// registered regardless — it simply does nothing until this is true — so a UART
|
||||
/// that only `reconfigure` discovers (via SPCR) still starts logging.
|
||||
pub fn present() bool {
|
||||
return uart_present;
|
||||
}
|
||||
|
||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||
@@ -70,15 +104,19 @@ pub fn reconfigure(is_mmio: bool, address: u64) void {
|
||||
}
|
||||
|
||||
fn writeByte(c: u8) void {
|
||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
||||
// Wait for the transmit-holding register to empty. `write` only reaches here
|
||||
// for a UART the loopback probe proved live, so this bounds a momentary stall
|
||||
// (e.g. deasserted flow control), not an absent port: ~5000 legacy-port reads
|
||||
// is a few ms — comfortably longer than one 38400-baud byte-time (~260 µs).
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
||||
while (register(5) & 0x20 == 0 and guard < 5_000) : (guard += 1) {}
|
||||
setRegister(0, c);
|
||||
}
|
||||
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up. A no-op
|
||||
/// when no working UART was detected, so a dead COM1 costs nothing per byte.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
if (!uart_present) return;
|
||||
for (bytes) |c| {
|
||||
if (c == '\n') writeByte('\r');
|
||||
writeByte(c);
|
||||
|
||||
@@ -148,7 +148,14 @@ pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3:
|
||||
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
||||
const deadline = apic.millis() + 100;
|
||||
while (apic.millis() < deadline) {
|
||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
|
||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) {
|
||||
// Cross-check this core's TSC against the BSP's before it joins the run
|
||||
// loop: an unsynchronized TSC must be caught before any task can migrate
|
||||
// onto this core and observe time going backward. No-op unless the TSC is
|
||||
// the clocksource (apic.checkWarpSource).
|
||||
apic.checkWarpSource();
|
||||
return true;
|
||||
}
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return false;
|
||||
@@ -166,6 +173,7 @@ fn delayMicros(us: u64) void {
|
||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
const cpu = boot_index;
|
||||
paging.setupPat(); // this core's PAT: entry 4 = write-combining, to match the BSP
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
@@ -177,6 +185,11 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
|
||||
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||
|
||||
// Rendezvous with the BSP for the TSC warp check (no-op unless the TSC is the
|
||||
// clocksource) before joining the run loop, so this core's clock is vetted before
|
||||
// it can run any task.
|
||||
apic.checkWarpTarget();
|
||||
|
||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||
}
|
||||
|
||||
@@ -189,7 +189,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.Device
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
|
||||
// Idempotent on exact match (docs/m19-m20-plan.md decision 3): a restarted
|
||||
// Idempotent on exact match (docs/device-manager.md): a restarted
|
||||
// registering bus re-registers what it rediscovers, and the table has no
|
||||
// unregister — an identical (class, identity, resources) child under the
|
||||
// same parent returns the existing id instead of appending a duplicate.
|
||||
|
||||
@@ -37,7 +37,9 @@ const Task = scheduler.Task;
|
||||
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||
|
||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||
pub const maximum_services = 8;
|
||||
// The name registry is indexed directly by ServiceId, so this must exceed the
|
||||
// largest id (currently fat = 8). Sized with headroom for new services.
|
||||
pub const maximum_services = 16;
|
||||
|
||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
|
||||
+85
-41
@@ -5,6 +5,7 @@ const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const console = @import("console.zig");
|
||||
const log = @import("log.zig");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
@@ -59,16 +60,34 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||
// with port-0x80 checkpoints as the only progress signal.
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
//
|
||||
// Serial is compiled in only under -Dserial (build.zig): a real machine often
|
||||
// has no live legacy COM1, and the log survives in the RAM buffer (below) and
|
||||
// is flushed to disk — so serial is now a QEMU/dev convenience the flashable
|
||||
// image leaves out. When it *is* built in, `serialInit`'s loopback probe still
|
||||
// guards against a dead port (so a -Dserial image is safe on real hardware).
|
||||
if (build_options.serial) {
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
}
|
||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||
// see it on a headless/real machine with no host capturing serial.
|
||||
log.addSink(log.ramSink);
|
||||
|
||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||
//
|
||||
// The console is brought up *after* paging (below), not here: its one-time
|
||||
// full-screen clear then runs on the kernel's **write-combining** mapping of the
|
||||
// framebuffer instead of the loader's uncached one — a fast burst rather than
|
||||
// millions of uncached writes on real hardware. Until then, on-screen output is
|
||||
// absent (an early panic still lands in the serial/RAM log); the trade is worth
|
||||
// a near-instant boot. `console.write` is a safe no-op while the console is down.
|
||||
const fb = boot_information.framebuffer;
|
||||
console.init(fb);
|
||||
|
||||
log.checkpoint(cp_entry);
|
||||
|
||||
@@ -77,12 +96,12 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
architecture.setFaultHandler(onException);
|
||||
architecture.init();
|
||||
|
||||
status("danos: initialising kernel...\n");
|
||||
log.write(if (console.present())
|
||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
status("/system/kernel: initialising kernel...\n");
|
||||
if (build_options.serial) log.write(if (architecture.serialPresent())
|
||||
"/system/kernel: serial console online (COM1)\n"
|
||||
else
|
||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
||||
"/system/kernel: no serial UART (COM1 absent) -> log kept in RAM/debugcon\n");
|
||||
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
||||
@@ -105,7 +124,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
const total_bytes = total_pages * abi.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
log.write("\ndanos: physical memory\n");
|
||||
log.write("\n/system/kernel: physical memory\n");
|
||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||
@@ -119,7 +138,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
||||
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
|
||||
const s1 = pmm.stats();
|
||||
log.print("\ndanos: frame allocator online\n", .{});
|
||||
log.print("\n/system/kernel: frame allocator online\n", .{});
|
||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||
const f0 = pmm.alloc();
|
||||
const f1 = pmm.alloc();
|
||||
@@ -133,14 +152,23 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
||||
log.checkpoint(cp_paging);
|
||||
log.print("\ndanos: paging enabled\n", .{});
|
||||
log.print("\n/system/kernel: paging enabled\n", .{});
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Now on our own tables, the framebuffer window is write-combining: bring up
|
||||
// the on-screen console and clear it (a fast burst here, not the loader's
|
||||
// uncached crawl). From here `status` reaches the screen as well as the log.
|
||||
console.init(fb);
|
||||
log.write(if (console.present())
|
||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
else
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
log.checkpoint(cp_heap);
|
||||
log.write("\ndanos: kernel heap online\n");
|
||||
log.write("\n/system/kernel: kernel heap online\n");
|
||||
// Measure the amount of resources the kernel is actually using
|
||||
const s2 = pmm.stats();
|
||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||
@@ -156,7 +184,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
};
|
||||
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
||||
var device_tree = devtree;
|
||||
log.write("\ndanos: device discovery online\n");
|
||||
log.write("\n/system/kernel: device discovery online\n");
|
||||
device_tree.dump(log.write);
|
||||
|
||||
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||
@@ -164,28 +192,20 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
devices_broker.init(&device_tree);
|
||||
if (devices_broker.dropped > 0) {
|
||||
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||
log.print("/system/kernel: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||
const pw = platform.powerInformation();
|
||||
log.write("danos: power\n");
|
||||
log.write("/system/kernel: power\n");
|
||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
} else {
|
||||
log.write(" S5 slp_typ : (not found)\n");
|
||||
}
|
||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||
|
||||
// AML namespace parse integrity: consumed should equal total.
|
||||
const am = platform.amlStats();
|
||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||
|
||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||
@@ -221,7 +241,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
});
|
||||
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
||||
|
||||
log.write("danos: platform\n");
|
||||
log.write("/system/kernel: platform\n");
|
||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
||||
@@ -237,7 +257,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
} else |err| {
|
||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
log.checkpoint(cp_discovery);
|
||||
|
||||
@@ -248,19 +268,43 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
scheduler.init(4);
|
||||
log.checkpoint(cp_scheduler);
|
||||
log.write("\ndanos: scheduler online\n");
|
||||
log.write("\n/system/kernel: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
architecture.startTimer();
|
||||
architecture.enableInterrupts();
|
||||
log.checkpoint(cp_timer);
|
||||
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||
log.print("/system/kernel: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||
|
||||
// The tsc-sync test forces the TSC clocksource on before the cores come up, so the
|
||||
// TSC + warp-check path is exercised even under TCG (which won't advertise an
|
||||
// invariant TSC). Inert in a normal build (docs/timers.md).
|
||||
if (build_options.test_case) |tc| {
|
||||
if (std.mem.eql(u8, tc, "tsc-sync")) architecture.forceTscClocksourceForTest();
|
||||
}
|
||||
|
||||
// Wake the other cores (application processors). A no-op on a single-core
|
||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). The
|
||||
// per-core TSC warp check rides this: each AP is vetted before it joins the run
|
||||
// loop (docs/timers.md).
|
||||
bringUpSecondaries();
|
||||
|
||||
// Report the monotonic clock's final reliability, now the warp check has run on
|
||||
// every core. On real Intel/AMD this is the invariant, synchronized TSC; a bare
|
||||
// VM (no invariant bit) or a machine whose cores' TSCs skew uses the HPET instead.
|
||||
log.print("/system/kernel: clocksource {s} (TSC invariant: {s}, synchronized: {s})\n", .{
|
||||
architecture.clockSourceName(),
|
||||
if (architecture.clockInvariant()) "yes" else "no",
|
||||
if (architecture.clockSynchronized()) "yes" else "no",
|
||||
});
|
||||
if (!architecture.clockSynchronized())
|
||||
log.write("/system/kernel: WARNING: per-core TSCs are not synchronized; monotonic clock moved off the TSC\n");
|
||||
|
||||
// Anchor wall-clock time: read the RTC once, now the monotonic clock is final.
|
||||
wall_clock.init();
|
||||
log.print("/system/kernel: wall clock {d} (Unix epoch seconds, UTC, from the RTC)\n", .{wall_clock.nowSeconds()});
|
||||
|
||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||
// Normal builds fall through to the idle halt.
|
||||
if (build_options.test_case) |case| {
|
||||
@@ -269,7 +313,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
}
|
||||
|
||||
log.checkpoint(cp_running);
|
||||
status("kernel initialised.\n");
|
||||
status("/system/kernel: initialised.\n");
|
||||
|
||||
// Publish the initial-ramdisk so user space can `system_spawn` its bundled
|
||||
// binaries by name. The kernel no longer launches them itself: init is the
|
||||
@@ -282,10 +326,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// manager then discovers the hardware and spawns each driver. init runs on its own
|
||||
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("starting /system/services/init...\n");
|
||||
status("/system/kernel: starting /system/services/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
||||
statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
statusPrint("/system/kernel: /system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /system/services/init on the boot volume.\n");
|
||||
@@ -294,7 +338,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
scheduler.setPriority(0);
|
||||
status("\nkernel idle; user space is running.\n");
|
||||
status("\n/system/kernel: kernel idle; user space is running.\n");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
@@ -321,7 +365,7 @@ fn bringUpSecondaries() void {
|
||||
// vector addresses it). It's kept for the system's life — armed only during a
|
||||
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
|
||||
if (ap_trampoline_page == 0) {
|
||||
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||
log.write("/system/kernel: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||
return;
|
||||
}
|
||||
architecture.setTrampolinePage(ap_trampoline_page);
|
||||
@@ -333,7 +377,7 @@ fn bringUpSecondaries() void {
|
||||
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
||||
}
|
||||
|
||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
log.print("\n/system/kernel: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
||||
for (cores[1..], 1..) |core, index| {
|
||||
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
||||
@@ -344,7 +388,7 @@ fn bringUpSecondaries() void {
|
||||
// This core's dedicated fault stack — allocated only now that the core is
|
||||
// real, rather than reserved statically for every possible core.
|
||||
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
||||
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||
log.print("/system/kernel: cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||
continue;
|
||||
};
|
||||
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||
@@ -353,14 +397,14 @@ fn bringUpSecondaries() void {
|
||||
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
||||
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||
pc.online = true;
|
||||
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||
log.print("/system/kernel: cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||
break;
|
||||
}
|
||||
if (attempt == maximum_wake_attempts)
|
||||
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||
log.print("/system/kernel: cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||
}
|
||||
}
|
||||
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
}
|
||||
|
||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||
@@ -427,7 +471,7 @@ fn exitReasonForVector(vector: u64) abi.ExitReason {
|
||||
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
statusPrint("\n/system/kernel: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
|
||||
@@ -41,6 +41,39 @@ pub fn write(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
|
||||
// --- the RAM sink: a retained copy of the whole diagnostic stream ------------
|
||||
//
|
||||
// A fixed in-image buffer that accumulates every logged byte, so a user program
|
||||
// (`log-flush`, and init at shutdown) can read it back through `klog_read` and
|
||||
// persist it to a file — the boot log survives on a headless/real machine that
|
||||
// has no host capturing serial. It is a *sink like any other*: register it with
|
||||
// `addSink(ramSink)` at boot. No allocation (works pre-heap and in a panic).
|
||||
//
|
||||
// It fills linearly and stops when full: the earliest output — the most valuable
|
||||
// for diagnosing a boot — is kept, and the tail is still on the live serial sink.
|
||||
// 256 KiB comfortably holds a full boot plus a long run (a boot is ~15 KiB).
|
||||
|
||||
const ram_capacity = 256 * 1024;
|
||||
var ram_buffer: [ram_capacity]u8 = undefined;
|
||||
var ram_len: usize = 0;
|
||||
|
||||
/// The RAM sink. Best-effort and self-guarding like every sink: appends what fits
|
||||
/// and silently drops the rest once full. (Concurrency matches the other sinks —
|
||||
/// the dominant writer, debug_write, already holds the kernel lock; a rare torn
|
||||
/// append on a kernel-internal line is an accepted diagnostic imperfection.)
|
||||
pub fn ramSink(bytes: []const u8) void {
|
||||
const n = @min(ram_buffer.len - ram_len, bytes.len);
|
||||
if (n != 0) {
|
||||
@memcpy(ram_buffer[ram_len..][0..n], bytes[0..n]);
|
||||
ram_len += n;
|
||||
}
|
||||
}
|
||||
|
||||
/// The accumulated log so far — what `klog_read` copies out.
|
||||
pub fn ramSnapshot() []const u8 {
|
||||
return ram_buffer[0..ram_len];
|
||||
}
|
||||
|
||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||
/// this is safe to call from interrupt context and from a panic.
|
||||
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||
|
||||
@@ -34,6 +34,7 @@ const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const log = @import("log.zig");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
const SystemCall = abi.SystemCall;
|
||||
@@ -206,6 +207,8 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.signal_bind => systemSignalBind(state),
|
||||
.process_signal => systemProcessSignal(state),
|
||||
.timer_bind => systemTimerBind(state),
|
||||
.klog_read => systemKlogRead(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -974,6 +977,37 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// klog_read(offset, ptr, len) -> bytes copied: copy the kernel's in-memory
|
||||
/// diagnostic log (the RAM sink in log.zig) out to the user buffer at `ptr`,
|
||||
/// starting at `offset`. Returns the count copied — 0 once `offset` reaches the
|
||||
/// end — so a program reads the whole log by looping from 0 until it gets 0.
|
||||
///
|
||||
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
||||
/// but the copy runs kernel -> user. Written under the kernel lock so the source
|
||||
/// snapshot can't grow underneath the copy. A read-only diagnostic — it exposes
|
||||
/// only the log the kernel already broadcasts to serial, nothing else.
|
||||
fn systemKlogRead(state: *architecture.CpuState) void {
|
||||
const offset = architecture.systemCallArg(state, 0);
|
||||
const ptr = architecture.systemCallArg(state, 1);
|
||||
const len = architecture.systemCallArg(state, 2);
|
||||
// Confine the whole destination span to the user (low) half. `len <=
|
||||
// user_half_end - ptr` bounds the length without an overflowing add.
|
||||
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const snapshot = log.ramSnapshot();
|
||||
var n: usize = 0;
|
||||
if (offset < snapshot.len) {
|
||||
n = @min(len, snapshot.len - offset);
|
||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
||||
@memcpy(dest[0..n], snapshot[offset..][0..n]);
|
||||
}
|
||||
architecture.setSystemCallResult(state, n);
|
||||
} else {
|
||||
fail(state);
|
||||
}
|
||||
}
|
||||
|
||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||
@@ -1304,3 +1338,10 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
||||
fn systemClock(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, architecture.nanos());
|
||||
}
|
||||
|
||||
/// wall_clock() -> Unix epoch seconds (UTC). The RTC value, read at boot and offset
|
||||
/// by the monotonic clock (wall-clock.zig) — mechanism, not policy: calendars and
|
||||
/// timezones layer on top in user space. Needed for filesystem timestamps (mtime).
|
||||
fn systemWallClock(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, wall_clock.nowSeconds());
|
||||
}
|
||||
|
||||
+297
-234
@@ -14,6 +14,7 @@ const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const architecture = @import("architecture");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const platform = @import("platform");
|
||||
const pmm = @import("pmm.zig");
|
||||
@@ -66,6 +67,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
timer();
|
||||
} else if (eql(case, "clock")) {
|
||||
clock();
|
||||
} else if (eql(case, "wall-clock")) {
|
||||
wallClock();
|
||||
} else if (eql(case, "vmm")) {
|
||||
vmm();
|
||||
} else if (eql(case, "heap")) {
|
||||
@@ -102,6 +105,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
stressTest();
|
||||
} else if (eql(case, "smp-retry")) {
|
||||
smpRetryTest();
|
||||
} else if (eql(case, "tsc-sync")) {
|
||||
tscSyncTest();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
@@ -142,6 +147,12 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
driverRestartTest(boot_information);
|
||||
} else if (eql(case, "usb-report")) {
|
||||
usbReportTest(boot_information);
|
||||
} else if (eql(case, "usb-hid")) {
|
||||
usbHidTest(boot_information);
|
||||
} else if (eql(case, "usb-storage")) {
|
||||
usbStorageTest(boot_information);
|
||||
} else if (eql(case, "fat-mount")) {
|
||||
fatMountTest(boot_information);
|
||||
} else if (eql(case, "device-list")) {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
@@ -152,26 +163,26 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
acpiReportTest(boot_information);
|
||||
} else if (eql(case, "acpi-ps2")) {
|
||||
acpiReportTest(boot_information); // same spawn; the harness regex differs
|
||||
} else if (eql(case, "power-button")) {
|
||||
acpiReportTest(boot_information); // boot the manager (spawns the acpi service); harness injects the button
|
||||
} else if (eql(case, "orderly-shutdown")) {
|
||||
orderlyShutdownTest(boot_information);
|
||||
} else if (eql(case, "initial-ramdisk")) {
|
||||
initialRamdiskTest(boot_information);
|
||||
} else if (eql(case, "vfs")) {
|
||||
vfsTest(boot_information);
|
||||
} else if (eql(case, "input")) {
|
||||
inputTest(boot_information);
|
||||
} else if (eql(case, "hpet")) {
|
||||
hpetTest(boot_information);
|
||||
} else if (eql(case, "iopass")) {
|
||||
ioPassTest();
|
||||
} else if (eql(case, "irqfree")) {
|
||||
irqFreeTest();
|
||||
} else if (eql(case, "bus")) {
|
||||
busTest(boot_information);
|
||||
} else if (eql(case, "containment")) {
|
||||
containmentTest();
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "poweroff")) {
|
||||
powerTest(.off);
|
||||
} else if (eql(case, "reboot")) {
|
||||
powerTest(.reboot);
|
||||
rebootTest();
|
||||
} else {
|
||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||
}
|
||||
@@ -188,15 +199,14 @@ fn platformHal() platform.Hal {
|
||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
||||
/// transition failed and we emit a FAIL result.
|
||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
||||
const name = if (action == .off) "poweroff" else "reboot";
|
||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
||||
// Soft-off (S5) is no longer a kernel operation — the ring-3 acpi service owns it
|
||||
// (exercised end-to-end by `orderly-shutdown`). Reboot stays in the kernel (FADT
|
||||
// reset register, no AML), so it keeps its own case.
|
||||
fn rebootTest() void {
|
||||
log("DANOS-TEST-BEGIN: reboot\n", .{});
|
||||
const hal = platformHal();
|
||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
||||
switch (action) {
|
||||
.off => platform.shutdown(hal),
|
||||
.reboot => platform.reboot(hal),
|
||||
}
|
||||
log("DANOS-POWER: attempting reboot\n", .{});
|
||||
platform.reboot(hal);
|
||||
check("power transition took effect", false);
|
||||
result();
|
||||
}
|
||||
@@ -207,6 +217,13 @@ fn eql(a: []const u8, b: []const u8) bool {
|
||||
return std.mem.eql(u8, a, b);
|
||||
}
|
||||
|
||||
/// Whether the captured last-write buffer *contains* `needle`. Markers are
|
||||
/// matched as substrings, not prefixes, so a service's source-path debug prefix
|
||||
/// (`system/drivers/pci-bus: ...`) still satisfies a marker like `pci-bus: `.
|
||||
fn bufferHas(needle: []const u8) bool {
|
||||
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
||||
}
|
||||
|
||||
/// Non-destructive checks of the memory map and frame allocator.
|
||||
fn smoke(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
||||
@@ -256,22 +273,19 @@ fn timer() void {
|
||||
}
|
||||
|
||||
/// Verify device discovery populated the platform facts the rest of the kernel
|
||||
/// depends on — the results ACPI parsing stashed in globals at boot. These are
|
||||
/// stable for the QEMU q35 + OVMF machine the harness runs, and span the tables:
|
||||
/// MADT (LAPIC base, CPU count), FADT (PM/reset registers), and the AML parse
|
||||
/// (the sleep type, plus the integrity check that every byte was consumed).
|
||||
/// depends on — the results the static ACPI tables stashed in globals at boot.
|
||||
/// These are stable for the QEMU q35 + OVMF machine the harness runs, and span the
|
||||
/// tables: MADT (LAPIC base, CPU count) and FADT (PM/reset registers). The kernel
|
||||
/// no longer interprets AML — sleep types are the ring-3 acpi service's concern.
|
||||
fn discoveryTest() void {
|
||||
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
||||
const pinfo = platform.platformInformation();
|
||||
const pw = platform.powerInformation();
|
||||
const am = platform.amlStats();
|
||||
|
||||
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
||||
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
||||
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
||||
check("reset register supported (FADT)", pw.reset_supported);
|
||||
check("S5 sleep type found (AML)", pw.s5 != null);
|
||||
check("AML parsed completely (consumed == total)", am.total > 0 and am.consumed == am.total);
|
||||
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
||||
|
||||
// M15: every PCI function now carries its own 4 KiB ECAM configuration space as
|
||||
@@ -441,6 +455,17 @@ fn heapTest() void {
|
||||
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
||||
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
||||
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
||||
fn wallClock() void {
|
||||
log("DANOS-TEST-BEGIN: wall-clock\n", .{});
|
||||
// The RTC was read and anchored at boot (kmain -> wall_clock.init()).
|
||||
const seconds = wall_clock.nowSeconds();
|
||||
log(" epoch: {d}\n", .{seconds});
|
||||
// A plausible current wall-clock: after 2020-01-01 (1577836800) and before 2050
|
||||
// (2524608000) — catches a broken CMOS read or a wrong epoch conversion.
|
||||
check("wall clock reads a plausible current epoch", seconds > 1_577_836_800 and seconds < 2_524_608_000);
|
||||
result();
|
||||
}
|
||||
|
||||
fn clock() void {
|
||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
||||
|
||||
@@ -1200,6 +1225,35 @@ fn clockTest() void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The TSC clocksource + cross-core warp check (`-smp 4`). This is the real
|
||||
/// Intel/AMD / KVM path — an invariant, synchronized TSC. TCG (the only x86
|
||||
/// accelerator on an Apple-Silicon host) won't advertise an invariant TSC, so the
|
||||
/// boot forces the TSC clocksource on (kernel.zig, gated on this case) to exercise
|
||||
/// the machinery: the kernel must run the per-AP warp check as each core came up,
|
||||
/// find the cores' TSCs synchronized, and keep the clock on the TSC (no HPET
|
||||
/// fallback). The default suite (no force) exercises the HPET fallback instead.
|
||||
fn tscSyncTest() void {
|
||||
log("DANOS-TEST-BEGIN: tsc-sync\n", .{});
|
||||
check("clocksource is the TSC (forced-invariant path)", eql(architecture.clockSourceName(), "tsc"));
|
||||
check("the cross-core warp check ran on the APs", architecture.warpChecksRun() >= 1);
|
||||
check("per-core TSCs synchronized (no warp, no HPET fallback)", architecture.clockSynchronized());
|
||||
|
||||
// A warp that slipped past the bring-up check would surface as a backward reading.
|
||||
var last = architecture.nanos();
|
||||
var monotonic = true;
|
||||
var advanced = false;
|
||||
var i: u32 = 0;
|
||||
while (i < 1_000_000) : (i += 1) {
|
||||
const t = architecture.nanos();
|
||||
if (t < last) monotonic = false;
|
||||
if (t > last) advanced = true;
|
||||
last = t;
|
||||
}
|
||||
check("monotonic clock advanced", advanced);
|
||||
check("monotonic clock never ran backward", monotonic);
|
||||
result();
|
||||
}
|
||||
|
||||
var proc_worker_run: bool = true;
|
||||
var proc_worker_ran: bool = false;
|
||||
|
||||
@@ -1376,7 +1430,7 @@ fn initTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const prefix = "init: heartbeat";
|
||||
const beat_ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
const beat_ok = bufferHas(prefix);
|
||||
check("init produced repeated heartbeats (>=2)", process.write_count >= 2);
|
||||
check("heartbeat text arrived intact", beat_ok);
|
||||
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -1638,11 +1692,11 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
var deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked)) break;
|
||||
if (bufferHas(parked)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("client parked holding an open handle", process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked));
|
||||
check("client parked holding an open handle", bufferHas(parked));
|
||||
|
||||
check("the kill is accepted", process.killProcess(me, client) == 0);
|
||||
var badge: u64 = 0;
|
||||
@@ -1655,11 +1709,11 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= released.len and eql(process.write_buffer[0..released.len], released)) break;
|
||||
if (bufferHas(released)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the VFS released the dead client's handle", process.write_len >= released.len and eql(process.write_buffer[0..released.len], released));
|
||||
check("the VFS released the dead client's handle", bufferHas(released));
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -1701,8 +1755,8 @@ fn signalsTest(boot_information: *const BootInformation) void {
|
||||
var saw_pass = false;
|
||||
var saw_fail = false;
|
||||
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||
if (process.write_len >= pass_marker.len and eql(process.write_buffer[0..pass_marker.len], pass_marker)) saw_pass = true;
|
||||
if (process.write_len >= fail_marker.len and eql(process.write_buffer[0..fail_marker.len], fail_marker)) saw_fail = true;
|
||||
if (bufferHas(pass_marker)) saw_pass = true;
|
||||
if (bufferHas(fail_marker)) saw_fail = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
@@ -1864,13 +1918,11 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
var deadline = architecture.millis() + 15000;
|
||||
while (architecture.millis() < deadline and reported == 0) {
|
||||
if (process.write_len > count_prefix.len + count_suffix.len and eql(process.write_buffer[0..count_prefix.len], count_prefix)) {
|
||||
const line = process.write_buffer[0..process.write_len];
|
||||
const digits_end = std.mem.indexOf(u8, line, count_suffix) orelse {
|
||||
scheduler.yield();
|
||||
continue;
|
||||
};
|
||||
reported = std.fmt.parseInt(u32, line[count_prefix.len..digits_end], 10) catch 0;
|
||||
const line = process.write_buffer[0..process.write_len];
|
||||
if (std.mem.indexOf(u8, line, count_prefix)) |start| {
|
||||
if (std.mem.indexOf(u8, line, count_suffix)) |digits_end| {
|
||||
reported = std.fmt.parseInt(u32, line[start + count_prefix.len .. digits_end], 10) catch 0;
|
||||
}
|
||||
}
|
||||
scheduler.yield();
|
||||
}
|
||||
@@ -1894,7 +1946,7 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
deadline = architecture.millis() + 15000;
|
||||
var restarted = false;
|
||||
while (architecture.millis() < deadline and !restarted) {
|
||||
if (process.write_len >= restart_marker.len and eql(process.write_buffer[0..restart_marker.len], restart_marker)) restarted = true;
|
||||
if (bufferHas(restart_marker)) restarted = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
@@ -1906,7 +1958,7 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
deadline = architecture.millis() + 15000;
|
||||
var seen = false;
|
||||
while (architecture.millis() < deadline and !seen) {
|
||||
if (process.write_len >= marker.len and eql(process.write_buffer[0..marker.len], marker)) seen = true;
|
||||
if (bufferHas(marker)) seen = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
@@ -1924,6 +1976,87 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// M21.3 capstone: orderly shutdown. Boot init with the initial-ramdisk
|
||||
/// published, so init spawns the full service tree (vfs, input, device-manager
|
||||
/// -> discovery/acpi); the harness injects a real power-button event via QMP;
|
||||
/// the acpi service publishes it; init runs the stop sequence over its children
|
||||
/// and asks the power service for S5; the machine powers off (QEMU exits). The
|
||||
/// kernel test only spawns init — the ordered chain is the harness assertion.
|
||||
/// The USB HID chain, end to end: boot the full service tree (init spawns vfs,
|
||||
/// input, device-manager), and let discovery run — the manager matches the PCI
|
||||
/// host bridge to pci-bus, pci-bus reports the xHCI controller, usb-xhci-bus
|
||||
/// enumerates the HID interfaces, and the manager spawns the class drivers. The
|
||||
/// harness's expect regex requires usb-xhci-bus to register the boot-keyboard
|
||||
/// interface, the manager to spawn usb-hid-keyboard, and that driver to come up
|
||||
/// (open its device, ask for boot protocol, subscribe) — proof the transfer
|
||||
/// protocol works class-driver to controller.
|
||||
fn usbHidTest(boot_information: *const BootInformation) void {
|
||||
bootServiceTreeTest(boot_information, "usb-hid");
|
||||
}
|
||||
|
||||
/// The USB storage chain: same full-tree boot, but the harness attaches a
|
||||
/// usb-storage device and the expect regex requires usb-storage to come up
|
||||
/// (open its device, run the BOT bring-up, read its capacity, and read block 0).
|
||||
fn usbStorageTest(boot_information: *const BootInformation) void {
|
||||
bootServiceTreeTest(boot_information, "usb-storage");
|
||||
}
|
||||
|
||||
/// The FAT mount chain: boot the full tree (init spawns the fat server, which
|
||||
/// brings up the USB storage chain, mounts the FAT volume, and mounts itself into
|
||||
/// the VFS at /mnt/usb), then spawn a fat-test client that lists and reads through
|
||||
/// the mount. The harness attaches a usb-storage device; the expect regex requires
|
||||
/// the fat mount and the client's success.
|
||||
fn fatMountTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: fat-mount\n", .{});
|
||||
if (boot_information.init_len == 0 or boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over init and the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
const init_ok = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false;
|
||||
check("init spawned (boots the tree, incl. the fat server)", init_ok);
|
||||
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
fn bootServiceTreeTest(boot_information: *const BootInformation, comptime label: []const u8) void {
|
||||
log("DANOS-TEST-BEGIN: " ++ label ++ "\n", .{});
|
||||
if (boot_information.init_len == 0 or boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over init and the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false;
|
||||
check("init spawned (boots vfs, input, device-manager, and the USB chain)", spawned);
|
||||
result();
|
||||
}
|
||||
|
||||
fn orderlyShutdownTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: orderly-shutdown\n", .{});
|
||||
if (boot_information.init_len == 0 or boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over init and the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false;
|
||||
check("init spawned as PID root of user space", spawned);
|
||||
result();
|
||||
}
|
||||
|
||||
/// M20.2: the acpi service registers + reports its _HID devices. Boot normally
|
||||
/// (the manager spawns discovery); the harness's expect regex requires the two
|
||||
/// PS/2 nodes among the service's report lines, each with its _CRS resources —
|
||||
@@ -1956,11 +2089,11 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// M20.1: the ring-3 AML parse agrees with the kernel's. The manager spawns
|
||||
/// the discovery service (the acpi build variant); it claims the acpi-tables
|
||||
/// node, maps the blobs, parses them, and logs its Device count — which must
|
||||
/// equal what the kernel's own parse produced (the equivalence that licenses
|
||||
/// retiring the kernel's device build in M20.3).
|
||||
/// M20.1: the ring-3 AML parse works. The manager spawns the discovery service
|
||||
/// (the acpi build variant); it claims the acpi-tables node, maps the blobs,
|
||||
/// parses them, and self-verifies it found at least a floor of Device objects,
|
||||
/// printing "acpi-parse: ok". The kernel no longer parses AML, so there is no
|
||||
/// kernel count to compare against — the ring-3 parse is now the only one.
|
||||
fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: acpi-parse\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
@@ -1975,23 +2108,18 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
||||
return;
|
||||
};
|
||||
|
||||
// The kernel's own count, from the namespace it already built for \_S5.
|
||||
const kernel_devices = platform.amlDeviceCount();
|
||||
check("the kernel namespace has devices to compare against", kernel_devices >= 1);
|
||||
|
||||
// Spawn the discovery service directly with that count as argv: it parses
|
||||
// the same blobs in ring 3 and self-verifies, printing "acpi-parse: ok" iff
|
||||
// the counts match. The harness's expect regex is that marker — deterministic,
|
||||
// no racing the shared serial buffer.
|
||||
// Spawn the discovery service directly with a device-count *floor* as argv:
|
||||
// it parses the blobs in ring 3 and self-verifies it found at least that many
|
||||
// Device objects, printing "acpi-parse: ok". A floor of 1 just proves the
|
||||
// parser ran and produced a namespace (the QEMU q35 DSDT has dozens). The
|
||||
// marker is deterministic — no racing the shared serial buffer.
|
||||
process.setInitialRamdisk(image);
|
||||
var count_text: [16]u8 = undefined;
|
||||
const count_arg = std.fmt.bufPrint(&count_text, "{d}", .{kernel_devices}) catch "0";
|
||||
var spawned = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "discovery")) continue;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", count_arg }, scheduler.currentId(), null) catch 0;
|
||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
|
||||
spawned = true;
|
||||
break;
|
||||
}
|
||||
@@ -2035,12 +2163,12 @@ fn supervisionTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= marker.len and eql(process.write_buffer[0..marker.len], marker)) break;
|
||||
if (bufferHas(marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= marker.len and eql(process.write_buffer[0..marker.len], marker);
|
||||
const ok = bufferHas(marker);
|
||||
if (!ok and process.write_len > 0) log("DANOS-SUPERVISION: got \"{s}\"\n", .{process.write_buffer[0..process.write_len]});
|
||||
check("the supervisor completed every step (spawn/list/kill/notify)", ok);
|
||||
check("it ran in user mode (CPL 3)", process.write_from_user);
|
||||
@@ -2120,12 +2248,12 @@ fn vfsTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
||||
if (bufferHas(prefix) and process.write_count >= 2) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
const ok = bufferHas(prefix);
|
||||
check("client completed the VFS round trip (open/write/read matched)", ok);
|
||||
check("the round trip ran repeatedly (server stays up)", process.write_count >= 2);
|
||||
check("client syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -2165,12 +2293,12 @@ fn inputTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 12000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
||||
if (bufferHas(prefix) and process.write_count >= 2) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
const ok = bufferHas(prefix);
|
||||
check("a subscriber received a broadcast key event over IPC (source -> service -> subscriber)", ok);
|
||||
check("events kept flowing (service + async send stay up)", process.write_count >= 2);
|
||||
check("client syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -2228,68 +2356,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
||||
return false;
|
||||
}
|
||||
|
||||
/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is
|
||||
/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its
|
||||
/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt
|
||||
/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken
|
||||
/// `target_ticks` times — it cannot reach that line by polling, because the loop's
|
||||
/// only exit is through `replyWait` returning a notification badge.
|
||||
///
|
||||
/// The interesting assertion is the last one, which doesn't trust hpet at all: it
|
||||
/// reads the I/O APIC's redirection entry back and checks the kernel really routed
|
||||
/// the line (our vector, level-triggered) and really left it unmasked after the
|
||||
/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so
|
||||
/// that state is quiescent and not a race.
|
||||
fn hpetTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: hpet\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet"));
|
||||
|
||||
const prefix = "hpet: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("user driver mapped HPET MMIO and was woken by its interrupt", ok);
|
||||
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk());
|
||||
result();
|
||||
}
|
||||
|
||||
/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the
|
||||
/// kernel programmed it: a vector in the device window, level-triggered, unmasked.
|
||||
/// Independent of anything the driver reported about itself.
|
||||
fn hpetRouteOk() bool {
|
||||
const gsi = hpetGsi() orelse return false;
|
||||
if (gsi >= architecture.irqRouteCount()) return false;
|
||||
const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0)
|
||||
const vector: u8 = @truncate(low & 0xFF);
|
||||
const masked = low & (1 << 16) != 0;
|
||||
const level = low & (1 << 15) != 0;
|
||||
return vector >= architecture.irq_vector_base and
|
||||
vector < architecture.irq_vector_base + architecture.irq_vector_count and
|
||||
level and !masked;
|
||||
}
|
||||
|
||||
/// The GSI discovery recorded for the HPET, from the same device table the driver saw.
|
||||
/// The GSI discovery recorded for the HPET, from the same device table drivers see.
|
||||
fn hpetGsi() ?u32 {
|
||||
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
@@ -2304,87 +2371,87 @@ fn hpetGsi() ?u32 {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Bus driver: a user process claims a device that contains other devices, enumerates
|
||||
/// them from the hardware, and publishes each as a child via `device_register` — the
|
||||
/// primitive a PCI bridge or USB hub driver is built from.
|
||||
///
|
||||
/// `bus` treats the HPET's register block as a bus and its comparators as children,
|
||||
/// giving each a 0x20 sub-window. It checks its own work (children come back from the
|
||||
/// table with the right parent and a strictly narrower window) and, importantly, that
|
||||
/// the kernel **refuses** a child whose window escapes the parent's — without that,
|
||||
/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints
|
||||
/// "bus: ok" only if all of that holds.
|
||||
///
|
||||
/// The kernel-side check here is the one bus can't make: that the children really did
|
||||
/// land in the device table with the containment invariant intact.
|
||||
fn busTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: bus\n", .{});
|
||||
/// `device_register` containment — a kernel security property, tested directly against
|
||||
/// the broker (no user-space demo driver). A bus driver publishes children of a device
|
||||
/// it owns; the kernel must **refuse** any child whose resource escapes the parent's
|
||||
/// grant, or `device_register` would become a system call for mapping arbitrary physical
|
||||
/// memory. This is the property the old `bus` demo driver proved end-to-end; with the
|
||||
/// demo gone, the property is asserted where it lives — in the kernel. Also checks the
|
||||
/// idempotence rule (M19.0): re-registering an identical child returns the same id
|
||||
/// instead of appending a duplicate.
|
||||
fn containmentTest() void {
|
||||
log("DANOS-TEST-BEGIN: containment\n", .{});
|
||||
|
||||
// M19.0: device_register is idempotent on exact match — a restarted
|
||||
// registering bus must not duplicate its children. Driven directly against
|
||||
// the broker: claim an unclaimed node, register the same (class, hid,
|
||||
// resourceless) child twice, expect one id and one table entry.
|
||||
{
|
||||
const me = scheduler.currentId();
|
||||
var probe: [1]device_abi.DeviceDescriptor = undefined;
|
||||
const total = devices_broker.enumerate(&probe);
|
||||
check("device tree is seeded for the idempotence check", total >= 1);
|
||||
if (devices_broker.ownerOf(0) == null) {
|
||||
check("claimed device 0 for the idempotence check", devices_broker.claim(0, me));
|
||||
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
||||
child.pci_class = device_abi.no_pci_class;
|
||||
child.hid_len = 4;
|
||||
child.hid[0..4].* = "idem".*;
|
||||
const first = devices_broker.register(0, me, &child) catch 0;
|
||||
check("first register succeeded", first != 0);
|
||||
const before = devices_broker.enumerate(&probe);
|
||||
const second = devices_broker.register(0, me, &child) catch 0;
|
||||
check("re-register returned the same id", second == first);
|
||||
check("re-register grew nothing", devices_broker.enumerate(&probe) == before);
|
||||
devices_broker.releaseAllOwnedBy(me);
|
||||
} else {
|
||||
check("device 0 unexpectedly claimed before the idempotence check", false);
|
||||
}
|
||||
}
|
||||
const me = scheduler.currentId();
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
check("device tree is seeded", devices_broker.enumerate(&buffer) >= 1);
|
||||
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
// The kernel-seeded HPET timer block is a device with a memory resource — a natural
|
||||
// parent to publish sub-window children under, as a PCI bridge or USB hub would.
|
||||
const parent_id = hpetDeviceId() orelse {
|
||||
check("found a device with a memory window to parent children under", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const parent = buffer[@intCast(parent_id)];
|
||||
var window: ?device_abi.ResourceDescriptor = null;
|
||||
for (0..parent.resource_count) |j| {
|
||||
if (parent.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) window = parent.resources[j];
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
const parent_window = window orelse {
|
||||
check("parent exposes a memory window", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus"));
|
||||
|
||||
const prefix = "bus: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
||||
scheduler.yield();
|
||||
if (devices_broker.ownerOf(parent_id) != null) {
|
||||
check("parent device was unclaimed at the start of the test", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("claimed the parent device", devices_broker.claim(parent_id, me));
|
||||
defer devices_broker.releaseAllOwnedBy(me);
|
||||
|
||||
// A child whose window lies inside the parent's is accepted.
|
||||
var fits = childDescriptor("cfit", parent_window.start, 0x20);
|
||||
const before = devices_broker.enumerate(&buffer);
|
||||
const good = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||
check("a contained child is registered", good != 0);
|
||||
check("the contained child was appended to the table", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
// A child whose window escapes the parent's is refused with NotContained.
|
||||
var escapes = childDescriptor("cesc", parent_window.start, parent_window.len + 0x1000);
|
||||
const refused = if (devices_broker.register(parent_id, me, &escapes)) |_| false else |err| err == error.NotContained;
|
||||
check("an out-of-window child is refused (NotContained)", refused);
|
||||
check("the refused child left the table unchanged", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
// Idempotent on exact match: re-registering the accepted child returns its id and
|
||||
// appends nothing (M19.0 — a restarted bus re-reports what it rediscovers).
|
||||
const again = devices_broker.register(parent_id, me, &fits) catch 0;
|
||||
check("re-registering an identical child returns the same id", again != 0 and again == good);
|
||||
check("re-registering grew nothing", devices_broker.enumerate(&buffer) == before + 1);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("bus driver published children and the kernel refused an out-of-window one", ok);
|
||||
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
check("every registered child is contained in its parent", childrenContained());
|
||||
result();
|
||||
}
|
||||
|
||||
/// A minimal child descriptor with one memory resource, for the containment test.
|
||||
fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescriptor {
|
||||
var child = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device_abi.DeviceClass.unknown);
|
||||
child.pci_class = device_abi.no_pci_class;
|
||||
child.hid_len = @intCast(hid.len);
|
||||
@memcpy(child.hid[0..hid.len], hid);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = @intFromEnum(device_abi.ResourceKind.memory), .start = start, .len = len };
|
||||
return child;
|
||||
}
|
||||
|
||||
/// The device manager (a ring-3 service) enumerates /system/devices, matches each
|
||||
/// device to a driver, and — eventually — spawns it. This increment only checks the
|
||||
/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves
|
||||
/// it, printing "device-manager: ok". It uses no special privilege — the same
|
||||
/// `device_enumerate` any process could call. (Spawning is the next increment.)
|
||||
/// device to a driver, and spawns it. Proof of the whole discover -> match -> spawn ->
|
||||
/// driver-up chain: boot only the device-manager; it must discover the PCI host bridge,
|
||||
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
|
||||
/// pci-bus must reach its own live marker. It uses no special privilege — the same
|
||||
/// `device_enumerate` any process could call.
|
||||
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
@@ -2400,70 +2467,65 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||
};
|
||||
|
||||
// Let `system_spawn` find bundled binaries by name (the normal boot path does
|
||||
// this too). Only the device-manager is spawned here — so if `hpet` runs at all,
|
||||
// it's because the manager discovered the timer, matched, and spawned it.
|
||||
// this too). Only the device-manager is spawned here — so if `pci-bus` runs at
|
||||
// all, it's because the manager discovered the PCI host bridge, matched, and
|
||||
// spawned it.
|
||||
process.setInitialRamdisk(image);
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
|
||||
|
||||
// End-to-end proof: the driver the manager spawned reaches its own live marker.
|
||||
// `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO,
|
||||
// binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after,
|
||||
// so it's race-free to poll for. Its arrival means the whole
|
||||
// discover -> match -> system_spawn -> driver-up chain worked.
|
||||
const prefix = "hpet: ok";
|
||||
// End-to-end proof, read from kernel state — not the racy last-write serial buffer,
|
||||
// since many services keep logging after pci-bus. The manager must discover the PCI
|
||||
// host bridge, match pci-bus, and spawn it, and pci-bus must come up: claim the
|
||||
// bridge, map its ECAM, and register the functions it enumerates as children in the
|
||||
// device tree.
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
var spawned = false;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
||||
if (processRunning("pci-bus")) spawned = true;
|
||||
if (spawned and pciFunctionsRegistered()) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("device manager matched the timer and system_spawn'd hpet, which came up", ok);
|
||||
check("device manager discovered the PCI host bridge and spawned pci-bus", spawned);
|
||||
check("pci-bus came up and registered the functions it enumerated", pciFunctionsRegistered());
|
||||
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
result();
|
||||
}
|
||||
|
||||
/// Every child `bus` registered must have each of its resources inside a parent
|
||||
/// resource of the same kind — the invariant `device_register` exists to maintain,
|
||||
/// checked from the kernel's own table rather than the driver's word for it.
|
||||
///
|
||||
/// Only *registered* children are checked, not the whole tree. Firmware topology is
|
||||
/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host
|
||||
/// bridge's `bus_range`, because a bus-number range isn't an address window.
|
||||
fn childrenContained() bool {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
|
||||
const bus_id = hpetDeviceId() orelse return false;
|
||||
const p = buffer[@intCast(bus_id)];
|
||||
|
||||
var children: usize = 0;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.parent != bus_id) continue;
|
||||
children += 1;
|
||||
for (0..d.resource_count) |i| {
|
||||
const r = d.resources[i];
|
||||
var ok = false;
|
||||
for (0..p.resource_count) |j| {
|
||||
const pr = p.resources[j];
|
||||
if (pr.kind != r.kind) continue;
|
||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
||||
if (pr.start == r.start) ok = true;
|
||||
} else if (r.len != 0 and r.start >= pr.start and
|
||||
r.start + r.len <= pr.start + pr.len) ok = true;
|
||||
}
|
||||
if (!ok) return false;
|
||||
}
|
||||
/// Whether a live task was spawned under `name` (its argv[0]) — read from the kernel
|
||||
/// task table, the same snapshot `process_enumerate` exposes.
|
||||
fn processRunning(name: []const u8) bool {
|
||||
var table: [64]abi.ProcessDescriptor = undefined;
|
||||
const total = scheduler.enumerate(&table);
|
||||
for (table[0..@min(total, table.len)]) |d| {
|
||||
if (std.mem.eql(u8, d.name[0..d.name_length], name)) return true;
|
||||
}
|
||||
return children > 0; // bus must have published at least one
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Device id of the HPET (the bus bus claims), from the same table drivers see.
|
||||
/// Whether pci-bus registered at least one function under the PCI host bridge — proof
|
||||
/// it came up, claimed the bridge, mapped its ECAM, and walked configuration space.
|
||||
fn pciFunctionsRegistered() bool {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
var bridge_id: ?u64 = null;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) bridge_id = d.id;
|
||||
}
|
||||
const bid = bridge_id orelse return false;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.parent == bid) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Device id of the kernel-seeded HPET timer block (the node with a memory resource),
|
||||
/// from the same device table drivers see. Used as a containment-test parent.
|
||||
fn hpetDeviceId() ?u64 {
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
@@ -2481,8 +2543,9 @@ fn hpetDeviceId() ?u64 {
|
||||
/// (so a dead driver's device goes quiet instead of storming) and the slot cleared
|
||||
/// (so an ISR never posts a notification into the endpoint that is about to be freed).
|
||||
///
|
||||
/// This is the path `hpet` never takes — it runs forever — so it gets its own test.
|
||||
/// Two properties, both read back from the hardware rather than from our own state:
|
||||
/// A long-running driver that never exits wouldn't reach this teardown path, so it
|
||||
/// gets its own test that binds and releases directly. Two properties, both read back
|
||||
/// from the hardware rather than from our own state:
|
||||
///
|
||||
/// 1. A bound GSI is routed and unmasked.
|
||||
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
//! Wall-clock time: the CMOS real-time clock read once at boot and anchored to the
|
||||
//! monotonic clock, so a query is a cheap arithmetic offset — no per-call CMOS poll,
|
||||
//! no lock, no SMP hazard on the shared 0x70/0x71 ports.
|
||||
//!
|
||||
//! Wall-clock *seconds* are mechanism the kernel owns, exactly like the monotonic
|
||||
//! clock ([[time-architecture]]): reading the hardware's value is not policy.
|
||||
//! Calendars, timezones, and formatting layer on top in user space. It exists so the
|
||||
//! filesystem can stamp real timestamps (mtime) — see docs/zig-self-hosting.md.
|
||||
|
||||
const architecture = @import("architecture");
|
||||
|
||||
var boot_unix_seconds: u64 = 0;
|
||||
var boot_nanos: u64 = 0;
|
||||
|
||||
/// Read the RTC once and anchor it to the monotonic clock. Call at boot, after the
|
||||
/// monotonic clock is calibrated.
|
||||
pub fn init() void {
|
||||
boot_unix_seconds = architecture.readRtcUnixSeconds();
|
||||
boot_nanos = architecture.nanos();
|
||||
}
|
||||
|
||||
/// The current wall-clock time in Unix epoch seconds (UTC): the boot RTC value plus
|
||||
/// the monotonic time elapsed since. Zero until `init` runs.
|
||||
pub fn nowSeconds() u64 {
|
||||
return boot_unix_seconds + (architecture.nanos() -% boot_nanos) / 1_000_000_000;
|
||||
}
|
||||
@@ -17,10 +17,13 @@ pub const maximum_cpus = 128;
|
||||
|
||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP. Sized
|
||||
/// for the initial-ramdisk sweep (15 bundled binaries spawned at once) plus the
|
||||
/// for the initial-ramdisk sweep (the bundled binaries spawned at once) plus the
|
||||
/// device manager's supervised children with room to grow — at 16 the sweep
|
||||
/// started failing spawns once the bundle passed a dozen binaries.
|
||||
pub const maximum_tasks = 32;
|
||||
/// started failing spawns once the bundle passed a dozen binaries. Raised to 48
|
||||
/// for the USB stack: the xHCI bus driver spawns a supervised class-driver instance
|
||||
/// per matched interface (keyboard, mouse, mass storage), on top of the FAT and
|
||||
/// block servers and the growing ramdisk bundle.
|
||||
pub const maximum_tasks = 48;
|
||||
|
||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||
pub const kernel_stack_size = 16 * 1024;
|
||||
|
||||
+401
-64
@@ -1,22 +1,25 @@
|
||||
//! /system/services/acpi — the ACPI discovery service: the x86 firmware
|
||||
//! interpreter, moved out of ring 0 (docs/m19-m20-plan.md, M20). Claims the
|
||||
//! interpreter, moved out of ring 0 (docs/discovery.md). Claims the
|
||||
//! `acpi-tables` node the kernel publishes (the AML blobs, the broad io_port
|
||||
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
||||
//! ring 3 — the same parser and interpreter the kernel uses.
|
||||
//!
|
||||
//! M20.2 (this increment): after parsing, walk the namespace and, for each
|
||||
//! present Device with a hardware id (`_HID`), evaluate its current resource
|
||||
//! settings (`_CRS`) through a ring-3 `Hal` (port I/O over the claimed node),
|
||||
//! register it under the acpi-tables node (its I/O ports and IRQs contained by
|
||||
//! the node's broad grants), and report it to the device manager with its
|
||||
//! EISA-decoded hid as identity. Matching those reports to drivers (ps2-bus)
|
||||
//! and retiring the kernel's own device build follow in M20.3.
|
||||
//! It also owns the **event side** (M21): it registers the domain-named `.power`
|
||||
//! service, binds the SCI (System Control Interrupt), and on a power-button
|
||||
//! fixed event publishes `power_button` to subscribers — and on init's request
|
||||
//! writes S5 to power the machine off. The device discovery (M20) and the event
|
||||
//! handling both run in one `runtime.service.run` loop.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const aml = @import("aml");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const device = runtime.device;
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const power = runtime.power_protocol;
|
||||
/// AML opcode/prefix bytes by name (`zero_opcode`, `byte_prefix`, …) — so the `_HID`
|
||||
/// integer decode names the opcodes instead of bare 0x0A/0x0B/… (docs/coding-standards.md).
|
||||
const opcodes = aml.opcodes;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
@@ -27,6 +30,43 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
// window — the Hal routes every port access through this one claim.
|
||||
var node_id: u64 = 0;
|
||||
var io_resource_index: u64 = 0;
|
||||
// The SCI's irq resource index on the node (the len-1 irq, distinct from the
|
||||
// broad [0,256) window), for irqBind / irqAck.
|
||||
var sci_resource_index: u64 = 0;
|
||||
var has_sci = false;
|
||||
|
||||
// PM1 event/control and GPE register ports, read from the FADT copy the kernel
|
||||
// publishes on the node (M21). Port 0 means absent.
|
||||
var pm1a_evt: u16 = 0;
|
||||
var pm1b_evt: u16 = 0;
|
||||
var pm1_evt_len: u8 = 0;
|
||||
var pm1a_cnt: u16 = 0;
|
||||
var pm1b_cnt: u16 = 0;
|
||||
var gpe0_blk: u16 = 0;
|
||||
var gpe0_len: u8 = 0;
|
||||
var gpe1_blk: u16 = 0;
|
||||
var gpe1_len: u8 = 0;
|
||||
var smi_cmd: u16 = 0;
|
||||
var acpi_enable_value: u8 = 0;
|
||||
var s5_slp_typ_a: u8 = 0;
|
||||
var s5_slp_typ_b: u8 = 0;
|
||||
var s5_valid = false;
|
||||
|
||||
// PM1 event-register bits (ACPI): PWRBTN in the status/enable word is bit 8;
|
||||
// the control word's SCI_EN is bit 0; SLP_EN is bit 13.
|
||||
const pwrbtn_bit: u16 = 1 << 8;
|
||||
const sci_en_bit: u32 = 1 << 0;
|
||||
const slp_en: u32 = 1 << 13;
|
||||
|
||||
// The `.power` subscribers: endpoints handed over as capabilities, each
|
||||
// receiving events as buffered messages. Dropped on a failed send. The
|
||||
// subscriber's task id is kept too — a shutdown request is honored only from a
|
||||
// subscriber (init subscribes; a stray process does not), the soft gate that
|
||||
// stands in for "only the system supervisor may power off" without hardcoding
|
||||
// a pid the kernel's idle tasks would have taken.
|
||||
const maximum_subscribers = 8;
|
||||
var subscribers: [maximum_subscribers]?runtime.ipc.Handle = .{null} ** maximum_subscribers;
|
||||
var subscriber_tasks: [maximum_subscribers]u32 = .{0} ** maximum_subscribers;
|
||||
|
||||
// Pass-1 registration record (see main): what pass 2 reports.
|
||||
const Registered = struct { hid: [8]u8 = .{0} ** 8, hid_len: usize = 0, device_id: u64 = 0, resource_count: u64 = 0 };
|
||||
@@ -63,100 +103,366 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
||||
// own device count to self-verify against — deterministic, no log-scraping.
|
||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||
// devices is the check. Deterministic, no log-scraping.
|
||||
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("acpi: out of memory\n");
|
||||
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||
return;
|
||||
};
|
||||
const node = findTablesNode(buffer) orelse {
|
||||
_ = runtime.system.write("acpi: no acpi-tables node to claim\n");
|
||||
_ = runtime.system.write("/system/services/acpi: no acpi-tables node to claim\n");
|
||||
return;
|
||||
};
|
||||
node_id = node.id;
|
||||
if (!device.claim(node_id)) {
|
||||
_ = runtime.system.write("acpi: unable to claim acpi-tables\n");
|
||||
_ = runtime.system.write("/system/services/acpi: unable to claim acpi-tables\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Map each memory resource (an AML blob) and note the io_port resource.
|
||||
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
|
||||
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
|
||||
var blocks: [8][]const u8 = undefined;
|
||||
var block_count: usize = 0;
|
||||
var found_io = false;
|
||||
var fadt: ?[]const u8 = null;
|
||||
for (node.resources[0..@intCast(node.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.io_port) and !found_io) {
|
||||
io_resource_index = index;
|
||||
found_io = true;
|
||||
continue;
|
||||
}
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.irq) and resource.len == 1) {
|
||||
sci_resource_index = index;
|
||||
has_sci = true;
|
||||
continue;
|
||||
}
|
||||
if (resource.kind != @intFromEnum(device.ResourceKind.memory)) continue;
|
||||
const base = device.mmioMap(node_id, index) orelse continue;
|
||||
const pointer: [*]const u8 = @ptrFromInt(base);
|
||||
blocks[block_count] = pointer[0..@intCast(resource.len)];
|
||||
const bytes = pointer[0..@intCast(resource.len)];
|
||||
if (bytes.len >= 4 and std.mem.eql(u8, bytes[0..4], "FACP")) {
|
||||
fadt = bytes;
|
||||
continue;
|
||||
}
|
||||
if (block_count == blocks.len) continue;
|
||||
blocks[block_count] = bytes;
|
||||
block_count += 1;
|
||||
if (block_count == blocks.len) break;
|
||||
}
|
||||
if (block_count == 0) {
|
||||
_ = runtime.system.write("acpi: no AML blobs on the node\n");
|
||||
_ = runtime.system.write("/system/services/acpi: no AML blobs on the node\n");
|
||||
return;
|
||||
}
|
||||
|
||||
const result = aml.parse(runtime.allocator(), blocks[0..block_count]) catch {
|
||||
_ = runtime.system.write("acpi: AML parse failed\n");
|
||||
_ = runtime.system.write("/system/services/acpi: AML parse failed\n");
|
||||
return;
|
||||
};
|
||||
var namespace = result.namespace;
|
||||
const devices = aml.deviceCount(&namespace);
|
||||
writeLine("acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||
if (expected) |want| {
|
||||
if (devices == want) {
|
||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||
if (floor) |minimum| {
|
||||
if (devices >= minimum) {
|
||||
_ = runtime.system.write("acpi-parse: ok\n");
|
||||
} else {
|
||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
||||
writeLine("acpi-parse: too few (ring-3 {d} < floor {d})\n", .{ devices, minimum });
|
||||
}
|
||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
// Register + report the present _HID devices (M20.2).
|
||||
var arena = std.heap.ArenaAllocator.init(runtime.allocator());
|
||||
var interpreter = aml.Interpreter.init(&namespace, .{
|
||||
// Register + report the present _HID devices (M20), then set up the power
|
||||
// event side (M21), then serve — all in one harness loop. The interpreter
|
||||
// and namespace outlive this frame (static), so the harness callbacks can
|
||||
// reach them.
|
||||
interpreter_arena = std.heap.ArenaAllocator.init(runtime.allocator());
|
||||
persistent_namespace = namespace;
|
||||
global_interpreter = aml.Interpreter.init(&persistent_namespace, .{
|
||||
.mapMmio = halMapMmio,
|
||||
.pioRead = halPioRead,
|
||||
.pioWrite = halPioWrite,
|
||||
}, arena.allocator());
|
||||
}, interpreter_arena.allocator());
|
||||
|
||||
// Pass 1: register every present _HID device under acpi-tables, remembering
|
||||
// each (hid, device id). Pass 2: report them all. Registering before any
|
||||
// report reaches the manager means a driver it spawns on the first report
|
||||
// already sees the whole set (no keyboard-before-mouse race for ps2-bus).
|
||||
readFadt(fadt);
|
||||
s5_valid = readSleepS5(&persistent_namespace);
|
||||
|
||||
runtime.service.run(power.message_maximum, .{
|
||||
.service = .power,
|
||||
.init = onInit,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
// Static so the harness callbacks (which run after main's stack frame is gone)
|
||||
// can reach the namespace and interpreter.
|
||||
var persistent_namespace: aml.Namespace = undefined;
|
||||
var global_interpreter: aml.Interpreter = undefined;
|
||||
var interpreter_arena: std.heap.ArenaAllocator = undefined;
|
||||
|
||||
/// Startup under the harness: register + report the discovered devices to the
|
||||
/// manager (M20), then enable ACPI mode and arm the power button (M21).
|
||||
fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
registered_count = 0;
|
||||
walkDevices(namespace.root, &interpreter);
|
||||
walkDevices(persistent_namespace.root, &global_interpreter);
|
||||
|
||||
const manager = runtime.ipc.lookup(.device_manager);
|
||||
var i: usize = 0;
|
||||
while (i < registered_count) : (i += 1) {
|
||||
const entry = registered[i];
|
||||
writeLine("acpi: reported {s} (device {d}, {d} resources)\n", .{ entry.hid[0..entry.hid_len], entry.device_id, entry.resource_count });
|
||||
const hid = entry.hid[0..entry.hid_len];
|
||||
const desc = acpi_ids.description(hid);
|
||||
if (desc.len != 0)
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources) — {s}\n", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
else
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
||||
if (manager) |h| {
|
||||
var report = protocol.ChildAdded{
|
||||
.parent = node_id,
|
||||
.bus_address = entry.device_id,
|
||||
.identity = 0,
|
||||
.device_id = entry.device_id,
|
||||
};
|
||||
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||
}
|
||||
}
|
||||
writeLine("acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||
|
||||
// Stay resident: the claim holds, and the service is here to grow into the
|
||||
// supervised discoverer (M20.3, then the M21 event side on the SCI).
|
||||
while (true) runtime.system.sleep(1000);
|
||||
armPowerButton(endpoint);
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- power event side (M21) ---------------------------------------------------
|
||||
|
||||
/// Read the PM1 event/control and GPE register ports plus the SMI enable pair
|
||||
/// from the FADT copy on the node. Offsets are from the FADT table start (the
|
||||
/// SDT header is the first 36 bytes). Prefers the 32-bit port fields; QEMU's
|
||||
/// FADT populates them.
|
||||
fn readFadt(fadt: ?[]const u8) void {
|
||||
const f = fadt orelse {
|
||||
_ = runtime.system.write("acpi: no FADT on the node — power events off\n");
|
||||
return;
|
||||
};
|
||||
smi_cmd = @truncate(rd32(f, 48));
|
||||
acpi_enable_value = f[52];
|
||||
pm1a_evt = @truncate(rd32(f, 56));
|
||||
pm1b_evt = @truncate(rd32(f, 60));
|
||||
pm1a_cnt = @truncate(rd32(f, 64));
|
||||
pm1b_cnt = @truncate(rd32(f, 68));
|
||||
gpe0_blk = @truncate(rd32(f, 80));
|
||||
gpe1_blk = @truncate(rd32(f, 84));
|
||||
pm1_evt_len = if (f.len > 88) f[88] else 4;
|
||||
gpe0_len = if (f.len > 92) f[92] else 0;
|
||||
gpe1_len = if (f.len > 93) f[93] else 0;
|
||||
}
|
||||
|
||||
fn readSleepS5(ns: *aml.Namespace) bool {
|
||||
const st = aml.sleepState(ns, 5) orelse return false;
|
||||
s5_slp_typ_a = st.slp_typ_a;
|
||||
s5_slp_typ_b = st.slp_typ_b;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Enable ACPI mode if the firmware isn't already in it, then bind the SCI and
|
||||
/// set PWRBTN_EN so the power button raises an interrupt we can see.
|
||||
fn armPowerButton(endpoint: runtime.ipc.Handle) void {
|
||||
if (pm1a_cnt != 0 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0 and smi_cmd != 0) {
|
||||
// Switch to ACPI mode: write ACPI_ENABLE to the SMI command port, then
|
||||
// spin (bounded) until SCI_EN latches.
|
||||
halPioWrite(1, smi_cmd, acpi_enable_value);
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries += 1) {
|
||||
runtime.system.sleep(1);
|
||||
}
|
||||
}
|
||||
if (!has_sci) {
|
||||
_ = runtime.system.write("acpi: no SCI resource — power button unavailable\n");
|
||||
return;
|
||||
}
|
||||
if (!device.irqBind(node_id, sci_resource_index, endpoint)) {
|
||||
_ = runtime.system.write("acpi: SCI irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
// PWRBTN_EN lives in the PM1 enable register at evt_blk + evt_len/2.
|
||||
if (pm1a_evt != 0) {
|
||||
const en_port = pm1a_evt + pm1_evt_len / 2;
|
||||
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
|
||||
}
|
||||
if (pm1b_evt != 0) {
|
||||
const en_port = pm1b_evt + pm1_evt_len / 2;
|
||||
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
|
||||
}
|
||||
_ = runtime.system.write("acpi: power button armed\n");
|
||||
}
|
||||
|
||||
/// The SCI fired. Read PM1 status; a set PWRBTN_STS is the power button — clear
|
||||
/// it (write-1), publish, log. Any other set status is cleared and logged
|
||||
/// (GPE/Notify dispatch is M21.2). Always re-arm the line.
|
||||
fn onSci() void {
|
||||
var handled = false;
|
||||
inline for (.{ pm1a_evt, pm1b_evt }) |evt_port| {
|
||||
if (evt_port != 0) {
|
||||
const sts: u16 = @truncate(halPioRead(2, evt_port));
|
||||
if (sts & pwrbtn_bit != 0) {
|
||||
halPioWrite(2, evt_port, pwrbtn_bit); // write-1-to-clear
|
||||
handled = true;
|
||||
} else if (sts != 0) {
|
||||
halPioWrite(2, evt_port, sts); // clear whatever else latched
|
||||
}
|
||||
}
|
||||
}
|
||||
if (handled) {
|
||||
_ = runtime.system.write("power: button pressed\n");
|
||||
publishButton();
|
||||
}
|
||||
handleGpe();
|
||||
_ = device.irqAck(node_id, sci_resource_index);
|
||||
}
|
||||
|
||||
/// General-purpose events: for each set+enabled GPE bit, evaluate its `\_GPE`
|
||||
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
||||
/// method produced, and publish an event per notified device. Then clear the
|
||||
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
||||
/// host unit tests (docs/acpi.md — ACPI events); on real hardware it carries
|
||||
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
||||
fn handleGpe() void {
|
||||
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
||||
handleGpeBlock(gpe1_blk, gpe1_len, gpe0_len * 4);
|
||||
}
|
||||
|
||||
fn handleGpeBlock(blk: u16, len: u8, gpe_base: u32) void {
|
||||
if (blk == 0 or len == 0) return;
|
||||
const status_bytes = len / 2; // status half, then enable half
|
||||
var byte_index: u8 = 0;
|
||||
while (byte_index < status_bytes) : (byte_index += 1) {
|
||||
const sts: u8 = @truncate(halPioRead(1, blk + byte_index));
|
||||
const en: u8 = @truncate(halPioRead(1, blk + status_bytes + byte_index));
|
||||
const active = sts & en;
|
||||
if (active == 0) continue;
|
||||
var bit: u3 = 0;
|
||||
while (true) : (bit += 1) {
|
||||
if (active & (@as(u8, 1) << bit) != 0) {
|
||||
dispatchGpe(gpe_base + @as(u32, byte_index) * 8 + bit);
|
||||
}
|
||||
if (bit == 7) break;
|
||||
}
|
||||
halPioWrite(1, blk + byte_index, active); // write-1-to-clear the serviced bits
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate the `\_GPE._L%02X` or `_E%02X` handler for GPE number `n`, then
|
||||
/// publish an event for each device it notified.
|
||||
fn dispatchGpe(n: u32) void {
|
||||
const gpe_scope = aml.Namespace.resolve(&persistent_namespace, persistent_namespace.root, true, 0, &.{seg4("_GPE")}) orelse return;
|
||||
var name: [4]u8 = .{ '_', 'L', 0, 0 };
|
||||
writeHex2(name[2..4], n);
|
||||
var method = aml.Namespace.childOf(gpe_scope, name);
|
||||
if (method == null) {
|
||||
name[1] = 'E';
|
||||
method = aml.Namespace.childOf(gpe_scope, name);
|
||||
}
|
||||
const m = method orelse return; // no handler — the status bit was already cleared
|
||||
_ = global_interpreter.evaluate(m, &.{}) catch return;
|
||||
for (global_interpreter.takeNotifications()) |event| publishNotify(event.node, event.code);
|
||||
}
|
||||
|
||||
fn publishNotify(node: *aml.Node, code: u64) void {
|
||||
// Map the notified device's _HID to a domain event where we recognize it.
|
||||
var hid: [8]u8 = .{0} ** 8;
|
||||
if (readHid(node, &global_interpreter)) |h| hid = h;
|
||||
const which: power.Event = if (std.mem.eql(u8, hid[0..7], "PNP0C0A")) .battery else if (std.mem.eql(u8, hid[0..7], "ACPI0003")) .ac else if (std.mem.eql(u8, hid[0..7], "PNP0C0D")) .lid else .notify;
|
||||
var event = power.EventMessage{ .event = @intFromEnum(which), .code = @truncate(code) };
|
||||
event.hid = hid;
|
||||
writeLine("power: notify {s} code {d}\n", .{ hid[0..7], code });
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
/// Two lowercase hex digits of `n` into `out[0..2]`.
|
||||
fn writeHex2(out: []u8, n: u32) void {
|
||||
const digits = "0123456789ABCDEF";
|
||||
out[0] = digits[(n >> 4) & 0xF];
|
||||
out[1] = digits[n & 0xF];
|
||||
}
|
||||
|
||||
fn publishButton() void {
|
||||
const event = power.EventMessage{ .event = @intFromEnum(power.Event.power_button) };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
fn publishEvent(bytes: []const u8) void {
|
||||
for (&subscribers) |*slot| {
|
||||
if (slot.*) |handle| {
|
||||
if (!runtime.ipc.send(handle, bytes)) slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn isSubscriber(task: u32) bool {
|
||||
for (&subscribers, 0..) |*slot, si| {
|
||||
if (slot.* != null and subscriber_tasks[si] == task) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Enter S5 (soft off): write SLP_TYP|SLP_EN to the PM1 control register(s).
|
||||
/// Mirrors the kernel's power.zig sleepValue. Only reached from a PID-1
|
||||
/// shutdown request (M21.3).
|
||||
fn enterS5() void {
|
||||
if (!s5_valid or pm1a_cnt == 0) {
|
||||
_ = runtime.system.write("power: S5 unavailable\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("power: entering S5\n");
|
||||
halPioWrite(2, pm1a_cnt, (@as(u32, s5_slp_typ_a & 0x7) << 10) | slp_en);
|
||||
if (pm1b_cnt != 0) halPioWrite(2, pm1b_cnt, (@as(u32, s5_slp_typ_b & 0x7) << 10) | slp_en);
|
||||
// If control returns, the write did not take — say so instead of hanging.
|
||||
runtime.system.sleep(500);
|
||||
_ = runtime.system.write("power: S5 write did not take\n");
|
||||
}
|
||||
|
||||
// --- harness callbacks --------------------------------------------------------
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
// The only notification the service binds is the SCI (an IRQ badge).
|
||||
_ = badge;
|
||||
onSci();
|
||||
}
|
||||
|
||||
/// The `.power` protocol: subscribe (endpoint as the call's capability),
|
||||
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
|
||||
/// device manager's), so nothing here handles ChildAdded.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
if (message.len < 1) return 0;
|
||||
switch (message[0]) {
|
||||
@intFromEnum(power.Operation.subscribe) => {
|
||||
var status: i32 = -1;
|
||||
if (capability) |handle| {
|
||||
for (&subscribers, 0..) |*slot, si| {
|
||||
if (slot.* == null) {
|
||||
slot.* = handle;
|
||||
subscriber_tasks[si] = sender;
|
||||
status = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const r = power.Reply{ .status = status };
|
||||
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
|
||||
return @sizeOf(power.Reply);
|
||||
},
|
||||
@intFromEnum(power.Operation.shutdown) => {
|
||||
// Honored only from a power subscriber — init, which has already run
|
||||
// the stop sequence over everything else. The power service is
|
||||
// mechanism (write S5); deciding *when* to shut down and stopping
|
||||
// the rest of the system first is init's policy.
|
||||
const allowed = isSubscriber(sender);
|
||||
const r = power.Reply{ .status = if (allowed) 0 else -1 };
|
||||
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
|
||||
if (allowed) enterS5();
|
||||
return @sizeOf(power.Reply);
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Depth-first walk: register + report each present device with a _HID, then
|
||||
@@ -172,8 +478,10 @@ fn walkDevices(node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
|
||||
if (readHid(c, interpreter)) |hid| {
|
||||
// Skip PCI roots — pci-bus already reports PCI functions; ACPI adds
|
||||
// only the non-PCI _HID devices (docs/m19-m20-plan.md M20.2).
|
||||
if (!std.mem.eql(u8, hid[0..7], "PNP0A03") and !std.mem.eql(u8, hid[0..7], "PNP0A08")) {
|
||||
// only the non-PCI _HID devices (docs/device-manager.md — matching). The two
|
||||
// roots are named through the shared registry, not bare _HID strings.
|
||||
const id = acpi_ids.HardwareId.fromHid(hid[0..7]);
|
||||
if (id != .pci_bus and id != .pci_express_root_bridge) {
|
||||
registerDevice(c, hid, interpreter);
|
||||
}
|
||||
}
|
||||
@@ -192,7 +500,7 @@ fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) vo
|
||||
applyCrs(&descriptor, node, interpreter);
|
||||
|
||||
const id = device.register(node_id, &descriptor) orelse {
|
||||
writeLine("acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||
writeLine("/system/services/acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||
return;
|
||||
};
|
||||
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||
@@ -225,7 +533,9 @@ fn readHid(node: *aml.Node, interpreter: *aml.Interpreter) ?[8]u8 {
|
||||
if (hid.kind != .name or hid.value.len == 0) return null;
|
||||
const v = hid.value;
|
||||
switch (v[0]) {
|
||||
0x00, 0x01, 0xFF, 0x0A, 0x0B, 0x0C, 0x0E => {
|
||||
// A static _HID names an integer EISA id: Zero/One/Ones or a Byte/Word/DWord/
|
||||
// QWord integer prefix. Anything else is not an integer we can EISA-decode.
|
||||
opcodes.zero_opcode, opcodes.one_opcode, opcodes.ones_opcode, opcodes.byte_prefix, opcodes.word_prefix, opcodes.dword_prefix, opcodes.qword_prefix => {
|
||||
var p: usize = 0;
|
||||
const n = readIntObj(v, &p) orelse return null;
|
||||
_ = eisaIdToStr(@truncate(n), &buffer);
|
||||
@@ -237,6 +547,33 @@ fn readHid(node: *aml.Node, interpreter: *aml.Interpreter) ?[8]u8 {
|
||||
|
||||
// --- _CRS resource-template decode (ported from the kernel's acpi.zig) --------
|
||||
|
||||
/// A resource template is a byte list of descriptors. Each starts with a tag byte whose
|
||||
/// high bit picks the encoding: a *small* descriptor carries its type in bits [6:3] and
|
||||
/// its length in bits [2:0]; a *large* descriptor is the whole tag byte, followed by a
|
||||
/// 16-bit length. These are the descriptor types danos decodes into resources — named so
|
||||
/// the walk below reads by descriptor, not by 0x04/0x85/… (docs/coding-standards.md).
|
||||
const large_descriptor_bit: u8 = 0x80; // set in a tag byte => large descriptor
|
||||
const small_length_mask: u8 = 0x07; // low 3 bits of a small tag = body length
|
||||
const small_type_shift: u3 = 3; // small type sits in bits [6:3]
|
||||
|
||||
/// Small resource descriptor types (tag bits [6:3]). Non-exhaustive: an unhandled type
|
||||
/// is skipped by its length, not misread.
|
||||
const SmallResourceType = enum(u8) {
|
||||
irq = 0x04,
|
||||
io_port = 0x08,
|
||||
fixed_io_port = 0x09,
|
||||
end_tag = 0x0F,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Large resource descriptor types (the whole tag byte). Non-exhaustive for the same reason.
|
||||
const LargeResourceType = enum(u8) {
|
||||
memory32 = 0x85,
|
||||
memory32_fixed = 0x86,
|
||||
extended_irq = 0x89,
|
||||
_,
|
||||
};
|
||||
|
||||
fn applyCrs(descriptor: *device.DeviceDescriptor, node: *aml.Node, interpreter: *aml.Interpreter) void {
|
||||
const crs = aml.Namespace.childOf(node, seg4("_CRS")) orelse return;
|
||||
const obj = interpreter.evaluate(crs, &.{}) catch return;
|
||||
@@ -247,21 +584,21 @@ fn applyCrs(descriptor: *device.DeviceDescriptor, node: *aml.Node, interpreter:
|
||||
var i: usize = 0;
|
||||
while (i < bytes.len) {
|
||||
const tag = bytes[i];
|
||||
if (tag & 0x80 == 0) {
|
||||
const len: usize = tag & 0x07;
|
||||
if (tag & large_descriptor_bit == 0) {
|
||||
const len: usize = tag & small_length_mask;
|
||||
const body = i + 1;
|
||||
if (body + len > bytes.len) break;
|
||||
switch ((tag >> 3) & 0x0F) {
|
||||
0x04 => if (len >= 2) { // IRQ mask
|
||||
switch (@as(SmallResourceType, @enumFromInt((tag >> small_type_shift) & 0x0F))) {
|
||||
.irq => if (len >= 2) { // IRQ mask
|
||||
const mask = @as(u16, bytes[body]) | (@as(u16, bytes[body + 1]) << 8);
|
||||
var b: usize = 0;
|
||||
while (b < 16) : (b += 1) {
|
||||
if (mask & (@as(u16, 1) << @intCast(b)) != 0) addResource(descriptor, .irq, b, 1);
|
||||
}
|
||||
},
|
||||
0x08 => if (len >= 7) addResource(descriptor, .io_port, rd16(bytes, body + 1), bytes[body + 6]),
|
||||
0x09 => if (len >= 3) addResource(descriptor, .io_port, rd16(bytes, body), bytes[body + 2]),
|
||||
0x0F => break,
|
||||
.io_port => if (len >= 7) addResource(descriptor, .io_port, rd16(bytes, body + 1), bytes[body + 6]),
|
||||
.fixed_io_port => if (len >= 3) addResource(descriptor, .io_port, rd16(bytes, body), bytes[body + 2]),
|
||||
.end_tag => break,
|
||||
else => {},
|
||||
}
|
||||
i = body + len;
|
||||
@@ -270,10 +607,10 @@ fn applyCrs(descriptor: *device.DeviceDescriptor, node: *aml.Node, interpreter:
|
||||
const len: usize = @intCast(rd16(bytes, i + 1));
|
||||
const body = i + 3;
|
||||
if (body + len > bytes.len) break;
|
||||
switch (tag) {
|
||||
0x85 => if (len >= 17) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 13)),
|
||||
0x86 => if (len >= 9) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 5)),
|
||||
0x89 => if (len >= 2) {
|
||||
switch (@as(LargeResourceType, @enumFromInt(tag))) {
|
||||
.memory32 => if (len >= 17) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 13)),
|
||||
.memory32_fixed => if (len >= 9) addResource(descriptor, .memory, rd32(bytes, body + 1), rd32(bytes, body + 5)),
|
||||
.extended_irq => if (len >= 2) {
|
||||
const count = bytes[body + 1];
|
||||
var k: usize = 0;
|
||||
while (k < count and body + 2 + k * 4 + 4 <= body + len) : (k += 1) {
|
||||
@@ -325,22 +662,22 @@ fn readIntObj(bytes: []const u8, p: *usize) ?u64 {
|
||||
const op = bytes[p.*];
|
||||
p.* += 1;
|
||||
switch (op) {
|
||||
0x00 => return 0,
|
||||
0x01 => return 1,
|
||||
0xFF => return 1,
|
||||
0x0A => {
|
||||
opcodes.zero_opcode => return 0,
|
||||
opcodes.one_opcode => return 1,
|
||||
opcodes.ones_opcode => return 1,
|
||||
opcodes.byte_prefix => {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const v = bytes[p.*];
|
||||
p.* += 1;
|
||||
return v;
|
||||
},
|
||||
0x0B => {
|
||||
opcodes.word_prefix => {
|
||||
if (p.* + 2 > bytes.len) return null;
|
||||
const v = rd16(bytes, p.*);
|
||||
p.* += 2;
|
||||
return v;
|
||||
},
|
||||
0x0C => {
|
||||
opcodes.dword_prefix => {
|
||||
if (p.* + 4 > bytes.len) return null;
|
||||
const v = rd32(bytes, p.*);
|
||||
p.* += 4;
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
//! The block-device wire protocol — what a filesystem (the FAT server) says to a
|
||||
//! block driver (usb-storage) over its well-known `.block` endpoint. A protocol
|
||||
//! module like vfs-protocol / usb-transfer-protocol: extern-struct messages, an
|
||||
//! `Operation` tag, everything in one IPC message.
|
||||
//!
|
||||
//! Data path: read and write move whole blocks to or from a **caller-owned DMA
|
||||
//! buffer**, named by its physical address — the same physical-address handoff
|
||||
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||
//! reply headers do. (Safe while the IOMMU is unenforced; see docs/driver-model.md.)
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// geometry() -> { block_size, block_count }
|
||||
geometry = 0,
|
||||
/// read(lba, count, physical): read `count` blocks from `lba` into the buffer
|
||||
read = 1,
|
||||
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||
write = 2,
|
||||
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
reserved: u32 = 0,
|
||||
lba: u64,
|
||||
count: u32, // number of blocks (read/write)
|
||||
reserved2: u32 = 0,
|
||||
physical: u64, // caller's DMA buffer physical address (read/write)
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
block_size: u32, // geometry: bytes per block (512)
|
||||
reserved2: u32 = 0,
|
||||
block_count: u64, // geometry: total blocks; read/write: blocks moved
|
||||
};
|
||||
|
||||
pub const message_maximum: usize = 256;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
@@ -18,6 +18,8 @@
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const pci_class = @import("pci-class");
|
||||
const usb_ids = @import("usb-ids");
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const system = runtime.system;
|
||||
@@ -30,21 +32,14 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The driver that serves each device — the policy table. In a fuller system
|
||||
/// this comes from a manifest (docs/device-manager.md: the third bus type
|
||||
/// triggers it); for now a static map. `null` = no driver for this class yet.
|
||||
fn driverFor(d: device.DeviceDescriptor) ?[]const u8 {
|
||||
// The HPET timer node is still kernel-seeded (from the HPET table, not AML).
|
||||
// PS/2 and other _HID devices now arrive as acpi-service reports and match
|
||||
// in onChildAdded (M20.3), not from this boot snapshot.
|
||||
if (d.class == @intFromEnum(device.DeviceClass.timer)) return "hpet";
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller:
|
||||
/// Serial Bus Controller (0x0C) / USB Controller (0x03) / XHCI (0x30) — the names
|
||||
/// pci-class.zig decodes.
|
||||
const xhci_pci_class: u64 = 0x0C_03_30;
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||
const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.serial_bus),
|
||||
.subclass = @intFromEnum(pci_class.serial_bus.SubClass.usb),
|
||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||
});
|
||||
|
||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||
@@ -67,6 +62,36 @@ fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The driver that serves a *reported* USB interface by its (class, subclass,
|
||||
/// protocol) triple — the third bus after PCI and ACPI (docs/device-manager.md:
|
||||
/// matching stays code until the third bus). The xHCI bus driver reports each
|
||||
/// interface with this packed triple as its identity; the matched class driver is
|
||||
/// spawned with the interface's registered id as argv[1], which it presents to the
|
||||
/// bus driver to open the device.
|
||||
fn usbDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
const keyboard = comptime usb_ids.packTriple(
|
||||
@intFromEnum(usb_ids.Class.hid),
|
||||
@intFromEnum(usb_ids.hid.SubClass.boot),
|
||||
@intFromEnum(usb_ids.hid.Protocol.keyboard),
|
||||
);
|
||||
const mouse = comptime usb_ids.packTriple(
|
||||
@intFromEnum(usb_ids.Class.hid),
|
||||
@intFromEnum(usb_ids.hid.SubClass.boot),
|
||||
@intFromEnum(usb_ids.hid.Protocol.mouse),
|
||||
);
|
||||
const storage = comptime usb_ids.packTriple(
|
||||
@intFromEnum(usb_ids.Class.mass_storage),
|
||||
@intFromEnum(usb_ids.mass_storage.SubClass.scsi),
|
||||
@intFromEnum(usb_ids.mass_storage.Protocol.bulk_only),
|
||||
);
|
||||
return switch (identity) {
|
||||
keyboard => "usb-hid-keyboard",
|
||||
mouse => "usb-hid-mouse",
|
||||
storage => "usb-storage",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
/// Whether some driver entry already serves registered device `device_id` —
|
||||
/// a re-report after a bus restart must not spawn a second instance.
|
||||
fn driverForDevice(device_id: u64) bool {
|
||||
@@ -102,7 +127,7 @@ const Driver = struct {
|
||||
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||
device_id: u64 = protocol.no_device,
|
||||
// Whether this driver speaks the protocol (hello expected, deadline
|
||||
// enforced). Legacy drivers (hpet, ps2-bus) are supervised and restarted
|
||||
// enforced). Legacy drivers (e.g. ps2-bus) are supervised and restarted
|
||||
// but not yet required to hello.
|
||||
speaks_protocol: bool = false,
|
||||
process_id: u32 = 0,
|
||||
@@ -186,7 +211,7 @@ fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, report
|
||||
fn pruneChildrenOf(reporter: u32) void {
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.reporter == reporter) {
|
||||
writeLine("device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
@@ -233,7 +258,7 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
||||
spawnDriver(driver);
|
||||
return;
|
||||
}
|
||||
writeLine("device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||
writeLine("/system/services/device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||
}
|
||||
|
||||
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
||||
@@ -248,7 +273,7 @@ fn spawnDriver(driver: *Driver) void {
|
||||
argument_count = 1;
|
||||
}
|
||||
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||
writeLine("device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||
writeLine("/system/services/device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||
driver.state = .failed;
|
||||
return;
|
||||
};
|
||||
@@ -262,9 +287,9 @@ fn spawnDriver(driver: *Driver) void {
|
||||
driver.state = .running;
|
||||
}
|
||||
if (driver.device_id != protocol.no_device) {
|
||||
writeLine("device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||
writeLine("/system/services/device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||
} else {
|
||||
writeLine("device-manager: spawned {s}\n", .{driver.name()});
|
||||
writeLine("/system/services/device-manager: spawned {s}\n", .{driver.name()});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -276,7 +301,7 @@ fn onDriverExit(driver: *Driver) void {
|
||||
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
driver.state = .stopped;
|
||||
writeLine("device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||
writeLine("/system/services/device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const now = system.clock();
|
||||
@@ -284,13 +309,13 @@ fn onDriverExit(driver: *Driver) void {
|
||||
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
||||
if (driver.restarts >= crash_loop_cap) {
|
||||
driver.state = .failed;
|
||||
writeLine("device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||
writeLine("/system/services/device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
||||
driver.state = .restarting;
|
||||
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
||||
writeLine("device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
writeLine("/system/services/device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
||||
}
|
||||
|
||||
@@ -301,7 +326,7 @@ fn onDriverExit(driver: *Driver) void {
|
||||
fn sweepDeadlines() void {
|
||||
const now = system.clock();
|
||||
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
||||
writeLine("device-manager: test mode: killing the reporter\n", .{});
|
||||
writeLine("/system/services/device-manager: test mode: killing the reporter\n", .{});
|
||||
_ = system.kill(test_kill_pid);
|
||||
test_kill_pid = 0;
|
||||
}
|
||||
@@ -309,7 +334,7 @@ fn sweepDeadlines() void {
|
||||
if (!driver.used) continue;
|
||||
switch (driver.state) {
|
||||
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
||||
writeLine("device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||
writeLine("/system/services/device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||
_ = system.kill(driver.process_id);
|
||||
// The exit notification finishes the job via onDriverExit.
|
||||
},
|
||||
@@ -326,7 +351,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("device-manager: out of memory\n");
|
||||
_ = runtime.system.write("/system/services/device-manager: out of memory\n");
|
||||
return false;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
@@ -341,19 +366,15 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
addDriver("pci-bus", descriptor.id, true);
|
||||
continue;
|
||||
}
|
||||
// PCI functions no longer appear in the boot snapshot (M19.3): the
|
||||
// pci-bus driver reports them, and onChildAdded matches from reports.
|
||||
const driver_name = driverFor(descriptor) orelse continue;
|
||||
matched += 1;
|
||||
// Skip a singleton that is already alive (the initial-ramdisk sweep test
|
||||
// starts every bundled binary bare, this manager included) — spawning a
|
||||
// second instance would only lose the claim race and churn the log.
|
||||
if (!alreadySupervised(driver_name) and !system.isProcessRunning(driver_name)) {
|
||||
addDriver(driver_name, protocol.no_device, false);
|
||||
}
|
||||
// Nothing else is matched from the boot snapshot today. The kernel-seeded
|
||||
// HPET timer node is served by the kernel's own clock (docs/timers.md), not
|
||||
// a user-space driver; PCI functions and PS/2 _HID devices arrive later as
|
||||
// pci-bus / acpi-service reports and match in onChildAdded (docs/discovery.md).
|
||||
// A fuller system's static class->driver manifest (docs/device-manager.md)
|
||||
// would slot in here.
|
||||
}
|
||||
|
||||
// The discovery service (docs/m19-m20-plan.md M20): one per firmware, packed
|
||||
// The discovery service (docs/discovery.md): one per firmware, packed
|
||||
// under the neutral name "discovery", spawned once at startup. It finds and
|
||||
// claims the acpi-tables (or devicetree-blob) node itself. Not a per-device
|
||||
// match — it is the discoverer, not a driver bound to one device.
|
||||
@@ -367,9 +388,9 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
}
|
||||
|
||||
if (matched == 0) {
|
||||
_ = runtime.system.write("device-manager: no matchable devices\n");
|
||||
_ = runtime.system.write("/system/services/device-manager: no matchable devices\n");
|
||||
} else {
|
||||
_ = runtime.system.write("device-manager: ok\n");
|
||||
_ = runtime.system.write("/system/services/device-manager: ok\n");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -390,13 +411,13 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
var status: i32 = 0;
|
||||
if (hello.version != protocol.version) {
|
||||
status = -1;
|
||||
writeLine("device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||
writeLine("/system/services/device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||
} else if (driverByProcess(sender)) |driver| {
|
||||
driver.state = .running;
|
||||
writeLine("device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||
} else {
|
||||
status = -1;
|
||||
writeLine("device-manager: hello from unknown process {d}\n", .{sender});
|
||||
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||
}
|
||||
const hello_reply = protocol.HelloReply{ .status = status };
|
||||
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
||||
@@ -412,7 +433,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
var status: i32 = 0;
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||
writeLine("device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
writeLine("/system/services/device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
||||
// Matching from reports (M19.3): a registered child whose identity
|
||||
// names a driver gets one, once — re-reports after a bus restart
|
||||
@@ -421,6 +442,11 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
if (pciDriverForIdentity(report.identity)) |child_driver| {
|
||||
if (!driverForDevice(report.device_id)) addDriver(child_driver, report.device_id, true);
|
||||
}
|
||||
// USB interface match: the reported identity is the packed class triple,
|
||||
// and the class driver is spawned with the interface's registered id.
|
||||
if (usbDriverForIdentity(report.identity)) |usb_driver| {
|
||||
if (!driverForDevice(report.device_id)) addDriver(usb_driver, report.device_id, true);
|
||||
}
|
||||
// ACPI _HID match (M20.3): ps2-bus is a singleton that finds its own
|
||||
// devices by hid, so spawn it once, without a device assignment.
|
||||
const hid_len = std.mem.indexOfScalar(u8, &report.hid, 0) orelse report.hid.len;
|
||||
@@ -473,7 +499,7 @@ fn onChildRemoved(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
var status: i32 = -1;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
||||
writeLine("device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
status = 0;
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,108 @@
|
||||
//! system/services/fat/fat-test — a client that proves the FAT mount end to end:
|
||||
//! it waits for the fat server to mount the USB volume at /mnt/usb, lists the
|
||||
//! root directory through the VFS (which routes /mnt/usb to the fat backend), and
|
||||
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
||||
//! kernel test spawns it alongside init.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for /mnt/usb to be mounted — the fat server races us at boot (it must
|
||||
// bring up the whole USB storage chain first).
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory("/mnt/usb");
|
||||
if (opened == null) runtime.system.sleep(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = runtime.system.write("fat-test: /mnt/usb never became available\n");
|
||||
return;
|
||||
};
|
||||
|
||||
var count: u32 = 0;
|
||||
var entry: fs.Entry = .{};
|
||||
while (dir.next(&entry)) {
|
||||
writeLine("fat-test: entry '{s}' kind={d} size={d}\n", .{ entry.name(), @intFromEnum(entry.kind), entry.size });
|
||||
count += 1;
|
||||
if (count > 32) break;
|
||||
}
|
||||
dir.close();
|
||||
writeLine("fat-test: listed {d} entries\n", .{count});
|
||||
|
||||
// Read a known file off the boot volume through the mount (best effort): the
|
||||
// kernel image is an ELF, so its first bytes are the ELF magic.
|
||||
if (fs.open("/mnt/usb/system/kernel", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var magic: [4]u8 = undefined;
|
||||
const n = file.read(&magic) orelse 0;
|
||||
file.close();
|
||||
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
||||
_ = runtime.system.write("fat-test: read /mnt/usb/system/kernel ELF magic ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: /mnt/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
}
|
||||
}
|
||||
|
||||
// Exercise directory + file mutation through the mount: mkdir, create a file
|
||||
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
||||
if (fs.makeDirectory("/mnt/usb/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open("/mnt/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
// The created file carries a real modification time (stamped from the RTC).
|
||||
var mtime_ok = false;
|
||||
if (fs.attributes("/mnt/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
||||
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
||||
}
|
||||
if (mtime_ok) _ = runtime.system.write("fat-test: mtime ok\n");
|
||||
|
||||
// Rename it, then read from the new name and confirm the old name is gone.
|
||||
const renamed = fs.rename("/mnt/usb/TESTDIR/HELLO.TXT", "/mnt/usb/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/mnt/usb/TESTDIR/HELLO.TXT");
|
||||
if (renamed and old_gone) _ = runtime.system.write("fat-test: rename ok\n");
|
||||
var readback = false;
|
||||
if (fs.open("/mnt/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [16]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
||||
}
|
||||
const removed = fs.remove("/mnt/usb/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/mnt/usb/TESTDIR/RENAMED.TXT");
|
||||
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
||||
_ = runtime.system.write("fat-test: mutations ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
||||
}
|
||||
} else {
|
||||
_ = runtime.system.write("fat-test: mkdir /mnt/usb/TESTDIR failed\n");
|
||||
}
|
||||
|
||||
if (count > 0) {
|
||||
while (true) {
|
||||
_ = runtime.system.write("fat-test: ok\n");
|
||||
runtime.system.sleep(1000);
|
||||
}
|
||||
}
|
||||
_ = runtime.system.write("fat-test: root listing was empty\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,249 @@
|
||||
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||
//! the VFS at /mnt/usb. From then on the VFS forwards every open/read/write/
|
||||
//! status/readdir/close under /mnt/usb to this server, which serves the same
|
||||
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||
//!
|
||||
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||
//! block driver by physical address, and the engine copies sectors in and out of
|
||||
//! it.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const engine = @import("engine.zig");
|
||||
const on_disk = @import("on-disk.zig");
|
||||
const protocol = runtime.vfs_protocol;
|
||||
const dma = runtime.dma;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
const mount_point = "/mnt/usb";
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
const IpcBlock = struct {
|
||||
device: runtime.block.Device,
|
||||
bounce: dma.Region,
|
||||
|
||||
fn readBlock(context: *anyopaque, lba: u64, buffer: []u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (!self.device.read(lba, 1, self.bounce.physical)) return false;
|
||||
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(buffer[0..512], source[0..512]);
|
||||
return true;
|
||||
}
|
||||
fn writeBlock(context: *anyopaque, lba: u64, buffer: []const u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(destination[0..512], buffer[0..512]);
|
||||
if (!self.device.write(lba, 1, self.bounce.physical)) return false;
|
||||
device_dirty = true; // a block reached the device; a close will flush it
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
var ipc_block: IpcBlock = undefined;
|
||||
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
|
||||
// Open handles the VFS holds against this backend: each maps a node id to a
|
||||
// resolved engine node.
|
||||
const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn openAt(id: u64) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
return if (o.used) o else null;
|
||||
}
|
||||
|
||||
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
|
||||
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
|
||||
const n = @min(payload.len, out.len - protocol.reply_size);
|
||||
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
|
||||
return protocol.reply_size + n;
|
||||
}
|
||||
|
||||
fn fail(out: []u8) usize {
|
||||
return writeReply(out, .{ .status = -1 }, &.{});
|
||||
}
|
||||
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.system.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
const device = runtime.block.open() orelse {
|
||||
_ = runtime.system.write("/system/services/fat: no block device (no storage attached)\n");
|
||||
return false; // clean exit: nothing to serve
|
||||
};
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = runtime.system.write("/system/services/fat: block geometry unavailable\n");
|
||||
return false;
|
||||
};
|
||||
ipc_block = .{ .device = device, .bounce = dma.alloc(4096, dma.coherent) orelse return false };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
.context = &ipc_block,
|
||||
.block_size = geometry.block_size,
|
||||
.block_count = geometry.block_count,
|
||||
.readBlockFn = IpcBlock.readBlock,
|
||||
.writeBlockFn = IpcBlock.writeBlock,
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = runtime.system.write("/system/services/fat: not a FAT filesystem\n");
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/services/fat: mounted FAT ({s}, {d} clusters, partition lba {d})\n", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
// Mount ourselves into the VFS namespace at /mnt/usb (retry while the VFS
|
||||
// comes up). From here the VFS routes /mnt/usb/... to this server.
|
||||
var tries: u32 = 0;
|
||||
while (tries < 100) : (tries += 1) {
|
||||
if (runtime.fs.mount(mount_point, endpoint)) {
|
||||
writeLine("/system/services/fat: mounted {s}\n", .{mount_point});
|
||||
return true;
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
}
|
||||
_ = runtime.system.write("/system/services/fat: could not mount into the VFS\n");
|
||||
return true; // still serve directly, even if the namespace mount didn't take
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path into its parent directory and final component: "/a/b" -> ("/a",
|
||||
// "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
|
||||
var node = filesystem.resolve(path);
|
||||
if (node == null and flags & protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||
node = filesystem.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return fail(out);
|
||||
// O_TRUNC: replace an existing file's contents rather than overwriting in place
|
||||
// (frees the old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & protocol.truncate != 0 and !resolved.is_directory) {
|
||||
filesystem.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return fail(out);
|
||||
open_nodes[index] = .{ .used = true, .node = resolved };
|
||||
return writeReply(out, .{ .status = 0, .node = index }, &.{});
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = capability;
|
||||
_ = sender;
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
|
||||
// Stamp create/write with the current wall-clock time (mtime). Cheap, and it
|
||||
// keeps the engine pure (it takes the time as data, not a syscall).
|
||||
filesystem.current_time_epoch = runtime.system.wallClock();
|
||||
|
||||
switch (request.operation) {
|
||||
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags),
|
||||
.read => {
|
||||
const o = openAt(request.node) orelse return fail(out);
|
||||
var buffer: [protocol.maximum_payload]u8 = undefined;
|
||||
const want = @min(@as(usize, request.len), buffer.len);
|
||||
const n = filesystem.readFile(o.node, @intCast(request.offset), buffer[0..want]);
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, buffer[0..n]);
|
||||
},
|
||||
.write => {
|
||||
const o = openAt(request.node) orelse return fail(out);
|
||||
const data = payload[0..@min(payload.len, request.len)];
|
||||
const n = filesystem.writeFile(&o.node, @intCast(request.offset), data);
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
|
||||
},
|
||||
.status => {
|
||||
const o = openAt(request.node) orelse return fail(out);
|
||||
const kind: protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
const status = protocol.FileStatus{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime };
|
||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&status));
|
||||
},
|
||||
.readdir => {
|
||||
const o = openAt(request.node) orelse return fail(out);
|
||||
if (!o.node.is_directory) return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
|
||||
const listing = filesystem.listEntry(o.node, @intCast(request.offset)) orelse return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
|
||||
const kind: protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const header = protocol.DirectoryEntry{ .kind = @intFromEnum(kind), .name_len = @intCast(listing.name_len), .size = listing.size };
|
||||
var buffer: [protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(buffer[0..protocol.directory_entry_size], std.mem.asBytes(&header));
|
||||
const nlen = @min(listing.name_len, buffer.len - protocol.directory_entry_size);
|
||||
@memcpy(buffer[protocol.directory_entry_size..][0..nlen], listing.name_buffer[0..nlen]);
|
||||
const total = protocol.directory_entry_size + nlen;
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(total) }, buffer[0..total]);
|
||||
},
|
||||
.close => {
|
||||
if (openAt(request.node)) |o| o.used = false;
|
||||
// Durable-on-close: if any block reached the device since the last
|
||||
// flush, commit its cache to stable media now (best-effort). This is
|
||||
// what makes init's shutdown log flush survive a real power-off, and is
|
||||
// the right default for removable media the user may unplug.
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.mkdir => {
|
||||
const split = splitParent(payload[0..@min(payload.len, request.len)]);
|
||||
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||
if (filesystem.createDirectory(parent, split.leaf) == null) return fail(out);
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.unlink => {
|
||||
const split = splitParent(payload[0..@min(payload.len, request.len)]);
|
||||
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||
if (!filesystem.removeFile(parent, split.leaf)) return fail(out);
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.rename => {
|
||||
const both = payload[0..@min(payload.len, request.len)];
|
||||
const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out);
|
||||
const old_split = splitParent(both[0..sep]);
|
||||
const new_split = splitParent(both[sep + 1 ..]);
|
||||
// Same-directory rename only.
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return fail(out);
|
||||
const parent = filesystem.resolve(old_split.parent) orelse return fail(out);
|
||||
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return fail(out);
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
// A backend is never itself a mount target.
|
||||
.mount, .unmount => return fail(out),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .fat,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,310 @@
|
||||
//! The on-disk layout of a FAT filesystem — the boot sector / BIOS Parameter
|
||||
//! Block, directory entries, long-file-name entries, and the FAT32 FSInfo — as
|
||||
//! `align(1)` extern structs that bit-cast straight out of a 512-byte sector
|
||||
//! (multi-byte fields are little-endian, like usb-abi.zig). Pure data, plus the
|
||||
//! cluster-count FAT-type detection. Host-testable.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// The BIOS Parameter Block, common to FAT12/16/32 (offset 0..36 of the boot
|
||||
/// sector). The extended part that follows differs by FAT type.
|
||||
pub const BiosParameterBlock = extern struct {
|
||||
jump: [3]u8,
|
||||
oem_name: [8]u8,
|
||||
bytes_per_sector: u16 align(1),
|
||||
sectors_per_cluster: u8,
|
||||
reserved_sector_count: u16 align(1),
|
||||
fat_count: u8,
|
||||
root_entry_count: u16 align(1),
|
||||
total_sectors_16: u16 align(1),
|
||||
media: u8,
|
||||
fat_size_16: u16 align(1),
|
||||
sectors_per_track: u16 align(1),
|
||||
head_count: u16 align(1),
|
||||
hidden_sectors: u32 align(1),
|
||||
total_sectors_32: u32 align(1),
|
||||
};
|
||||
|
||||
/// The FAT12/16 extended boot record (offset 36).
|
||||
pub const ExtendedBootRecord16 = extern struct {
|
||||
drive_number: u8,
|
||||
reserved: u8,
|
||||
boot_signature: u8,
|
||||
volume_id: u32 align(1),
|
||||
volume_label: [11]u8,
|
||||
filesystem_type: [8]u8,
|
||||
};
|
||||
|
||||
/// The FAT32 extended boot record (offset 36).
|
||||
pub const ExtendedBootRecord32 = extern struct {
|
||||
fat_size_32: u32 align(1),
|
||||
extended_flags: u16 align(1),
|
||||
filesystem_version: u16 align(1),
|
||||
root_cluster: u32 align(1),
|
||||
filesystem_information_sector: u16 align(1),
|
||||
backup_boot_sector: u16 align(1),
|
||||
reserved: [12]u8,
|
||||
drive_number: u8,
|
||||
reserved1: u8,
|
||||
boot_signature: u8,
|
||||
volume_id: u32 align(1),
|
||||
volume_label: [11]u8,
|
||||
filesystem_type: [8]u8,
|
||||
};
|
||||
|
||||
/// A 32-byte directory entry (8.3 short name form).
|
||||
pub const DirectoryEntry = extern struct {
|
||||
name: [11]u8, // 8 name + 3 extension, space-padded
|
||||
attributes: u8,
|
||||
reserved_nt: u8,
|
||||
creation_time_tenth: u8,
|
||||
creation_time: u16 align(1),
|
||||
creation_date: u16 align(1),
|
||||
last_access_date: u16 align(1),
|
||||
first_cluster_high: u16 align(1),
|
||||
write_time: u16 align(1),
|
||||
write_date: u16 align(1),
|
||||
first_cluster_low: u16 align(1),
|
||||
file_size: u32 align(1),
|
||||
|
||||
pub fn firstCluster(self: DirectoryEntry) u32 {
|
||||
return (@as(u32, self.first_cluster_high) << 16) | self.first_cluster_low;
|
||||
}
|
||||
pub fn setFirstCluster(self: *DirectoryEntry, cluster: u32) void {
|
||||
self.first_cluster_low = @truncate(cluster);
|
||||
self.first_cluster_high = @truncate(cluster >> 16);
|
||||
}
|
||||
pub fn isFree(self: DirectoryEntry) bool {
|
||||
return self.name[0] == 0x00 or self.name[0] == 0xE5;
|
||||
}
|
||||
pub fn isEnd(self: DirectoryEntry) bool {
|
||||
return self.name[0] == 0x00;
|
||||
}
|
||||
pub fn isDirectory(self: DirectoryEntry) bool {
|
||||
return self.attributes & attribute_directory != 0;
|
||||
}
|
||||
pub fn isLongName(self: DirectoryEntry) bool {
|
||||
return self.attributes & attribute_long_name_mask == attribute_long_name;
|
||||
}
|
||||
pub fn isVolumeLabel(self: DirectoryEntry) bool {
|
||||
return self.attributes & attribute_volume_id != 0 and !self.isLongName();
|
||||
}
|
||||
};
|
||||
|
||||
/// A 32-byte long-file-name entry (attributes == 0x0F). A sequence of these
|
||||
/// precedes the 8.3 entry they name, each carrying 13 UTF-16 code units.
|
||||
pub const LongNameEntry = extern struct {
|
||||
order: u8,
|
||||
name1: [5]u16 align(1),
|
||||
attributes: u8,
|
||||
kind: u8,
|
||||
checksum: u8,
|
||||
name2: [6]u16 align(1),
|
||||
first_cluster_low: u16 align(1),
|
||||
name3: [2]u16 align(1),
|
||||
};
|
||||
|
||||
/// The FAT32 FSInfo sector (usually sector 1): advisory free-cluster bookkeeping.
|
||||
pub const FileSystemInformation = extern struct {
|
||||
lead_signature: u32 align(1), // 0x41615252
|
||||
reserved1: [480]u8,
|
||||
struct_signature: u32 align(1), // 0x61417272
|
||||
free_count: u32 align(1),
|
||||
next_free: u32 align(1),
|
||||
reserved2: [12]u8,
|
||||
trail_signature: u32 align(1), // 0xAA550000
|
||||
};
|
||||
|
||||
// Directory-entry attribute bits.
|
||||
pub const attribute_read_only: u8 = 0x01;
|
||||
pub const attribute_hidden: u8 = 0x02;
|
||||
pub const attribute_system: u8 = 0x04;
|
||||
pub const attribute_volume_id: u8 = 0x08;
|
||||
pub const attribute_directory: u8 = 0x10;
|
||||
pub const attribute_archive: u8 = 0x20;
|
||||
pub const attribute_long_name: u8 = 0x0F; // read_only|hidden|system|volume_id
|
||||
pub const attribute_long_name_mask: u8 = 0x3F;
|
||||
|
||||
// FSInfo signatures.
|
||||
pub const fsinfo_lead_signature: u32 = 0x41615252;
|
||||
pub const fsinfo_struct_signature: u32 = 0x61417272;
|
||||
pub const fsinfo_trail_signature: u32 = 0xAA550000;
|
||||
|
||||
/// End-of-chain markers (a cluster value >= these ends a chain).
|
||||
pub const end_of_chain_12: u32 = 0xFF8;
|
||||
pub const end_of_chain_16: u32 = 0xFFF8;
|
||||
pub const end_of_chain_32: u32 = 0x0FFFFFF8;
|
||||
pub const bad_cluster_32: u32 = 0x0FFFFFF7;
|
||||
|
||||
pub const free_cluster: u32 = 0;
|
||||
pub const boot_signature_offset: usize = 510; // 0x55 0xAA at the end of the boot sector
|
||||
|
||||
pub const FatType = enum { fat12, fat16, fat32 };
|
||||
|
||||
/// The geometry derived from the BPB, plus the FAT type (by the Microsoft
|
||||
/// cluster-count rule: <4085 FAT12, <65525 FAT16, else FAT32).
|
||||
pub const Geometry = struct {
|
||||
fat_type: FatType,
|
||||
bytes_per_sector: u32,
|
||||
sectors_per_cluster: u32,
|
||||
reserved_sector_count: u32,
|
||||
fat_count: u32,
|
||||
fat_size_sectors: u32, // per FAT
|
||||
root_entry_count: u32, // FAT12/16
|
||||
root_cluster: u32, // FAT32
|
||||
first_data_sector: u32,
|
||||
total_sectors: u32,
|
||||
cluster_count: u32,
|
||||
fsinfo_sector: u32, // FAT32
|
||||
};
|
||||
|
||||
/// Derive the geometry (and FAT type) from a boot sector's first 512 bytes.
|
||||
/// Returns null if the sector is not a plausible FAT boot sector.
|
||||
pub fn geometryOf(sector: []const u8) ?Geometry {
|
||||
if (sector.len < 512) return null;
|
||||
if (sector[boot_signature_offset] != 0x55 or sector[boot_signature_offset + 1] != 0xAA) return null;
|
||||
const bpb = std.mem.bytesToValue(BiosParameterBlock, sector[0..@sizeOf(BiosParameterBlock)]);
|
||||
if (bpb.bytes_per_sector == 0 or bpb.sectors_per_cluster == 0 or bpb.fat_count == 0) return null;
|
||||
|
||||
const fat_size_16: u32 = bpb.fat_size_16;
|
||||
var fat_size: u32 = fat_size_16;
|
||||
var root_cluster: u32 = 0;
|
||||
var fsinfo_sector: u32 = 0;
|
||||
if (fat_size_16 == 0) {
|
||||
const ebr = std.mem.bytesToValue(ExtendedBootRecord32, sector[36 .. 36 + @sizeOf(ExtendedBootRecord32)]);
|
||||
fat_size = ebr.fat_size_32;
|
||||
root_cluster = ebr.root_cluster;
|
||||
fsinfo_sector = ebr.filesystem_information_sector;
|
||||
}
|
||||
|
||||
const total_sectors: u32 = if (bpb.total_sectors_16 != 0) bpb.total_sectors_16 else bpb.total_sectors_32;
|
||||
const root_dir_sectors = (@as(u32, bpb.root_entry_count) * 32 + bpb.bytes_per_sector - 1) / bpb.bytes_per_sector;
|
||||
const first_data_sector = bpb.reserved_sector_count + bpb.fat_count * fat_size + root_dir_sectors;
|
||||
if (total_sectors < first_data_sector) return null;
|
||||
const data_sectors = total_sectors - first_data_sector;
|
||||
const cluster_count = data_sectors / bpb.sectors_per_cluster;
|
||||
|
||||
const fat_type: FatType = if (cluster_count < 4085) .fat12 else if (cluster_count < 65525) .fat16 else .fat32;
|
||||
|
||||
return .{
|
||||
.fat_type = fat_type,
|
||||
.bytes_per_sector = bpb.bytes_per_sector,
|
||||
.sectors_per_cluster = bpb.sectors_per_cluster,
|
||||
.reserved_sector_count = bpb.reserved_sector_count,
|
||||
.fat_count = bpb.fat_count,
|
||||
.fat_size_sectors = fat_size,
|
||||
.root_entry_count = bpb.root_entry_count,
|
||||
.root_cluster = root_cluster,
|
||||
.first_data_sector = first_data_sector,
|
||||
.total_sectors = total_sectors,
|
||||
.cluster_count = cluster_count,
|
||||
.fsinfo_sector = fsinfo_sector,
|
||||
};
|
||||
}
|
||||
|
||||
// --- DOS date/time <-> Unix epoch --------------------------------------------
|
||||
//
|
||||
// FAT stamps a file's modification time as two 16-bit DOS fields. There is no
|
||||
// timezone, so danos treats them as UTC. `date`: year-1980(7)|month(4)|day(5);
|
||||
// `time`: hour(5)|minute(6)|(second/2)(5).
|
||||
|
||||
fn isLeapYear(year: u32) bool {
|
||||
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0);
|
||||
}
|
||||
|
||||
const days_in_month = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||
|
||||
/// Convert a FAT date+time to Unix epoch seconds (UTC). Returns 0 for an unset
|
||||
/// (zero) date.
|
||||
pub fn fatToEpoch(date: u16, time: u16) u64 {
|
||||
if (date == 0) return 0;
|
||||
const day: u32 = date & 0x1F;
|
||||
const month: u32 = (date >> 5) & 0x0F;
|
||||
const year: u32 = 1980 + (date >> 9);
|
||||
if (month < 1 or month > 12 or day < 1) return 0;
|
||||
const second: u32 = @as(u32, time & 0x1F) * 2;
|
||||
const minute: u32 = (time >> 5) & 0x3F;
|
||||
const hour: u32 = (time >> 11) & 0x1F;
|
||||
|
||||
var days: u64 = 0;
|
||||
var y: u32 = 1970;
|
||||
while (y < year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||
var m: u32 = 1;
|
||||
while (m < month) : (m += 1) {
|
||||
days += days_in_month[m - 1];
|
||||
if (m == 2 and isLeapYear(year)) days += 1;
|
||||
}
|
||||
days += day - 1;
|
||||
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||
}
|
||||
|
||||
pub const FatDateTime = struct { date: u16, time: u16 };
|
||||
|
||||
/// Convert Unix epoch seconds (UTC) to a FAT date+time. Returns {0,0} for epoch 0 or
|
||||
/// any time before 1980 (which DOS cannot represent).
|
||||
pub fn epochToFatDateTime(epoch: u64) FatDateTime {
|
||||
if (epoch == 0) return .{ .date = 0, .time = 0 };
|
||||
var remaining = epoch;
|
||||
const second: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const minute: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const hour: u32 = @intCast(remaining % 24);
|
||||
remaining /= 24;
|
||||
var days: u32 = @intCast(remaining); // whole days since 1970-01-01
|
||||
|
||||
var year: u32 = 1970;
|
||||
while (true) {
|
||||
const y_days: u32 = if (isLeapYear(year)) 366 else 365;
|
||||
if (days < y_days) break;
|
||||
days -= y_days;
|
||||
year += 1;
|
||||
}
|
||||
if (year < 1980) return .{ .date = 0, .time = 0 };
|
||||
var month: u32 = 1;
|
||||
while (true) {
|
||||
var m_days: u32 = days_in_month[month - 1];
|
||||
if (month == 2 and isLeapYear(year)) m_days += 1;
|
||||
if (days < m_days) break;
|
||||
days -= m_days;
|
||||
month += 1;
|
||||
}
|
||||
const day = days + 1;
|
||||
return .{
|
||||
.date = @intCast(((year - 1980) << 9) | (month << 5) | day),
|
||||
.time = @intCast((hour << 11) | (minute << 5) | (second / 2)),
|
||||
};
|
||||
}
|
||||
|
||||
test "FAT date/time <-> Unix epoch round trip" {
|
||||
// Even-second UTC times (FAT stores seconds/2, so even seconds round-trip exactly).
|
||||
for ([_]u64{ 1_577_836_800, 1_700_000_000, 1_262_304_000, 1_783_971_244 }) |epoch| {
|
||||
const fat = epochToFatDateTime(epoch);
|
||||
try std.testing.expectEqual(epoch, fatToEpoch(fat.date, fat.time));
|
||||
}
|
||||
// Absolute check: 1577836800 is 2020-01-01 00:00:00 UTC.
|
||||
const y2020 = epochToFatDateTime(1_577_836_800);
|
||||
try std.testing.expectEqual(@as(u16, 2020), 1980 + (y2020.date >> 9));
|
||||
try std.testing.expectEqual(@as(u16, 1), (y2020.date >> 5) & 0x0F); // month
|
||||
try std.testing.expectEqual(@as(u16, 1), y2020.date & 0x1F); // day
|
||||
// 0 is "unset" both ways.
|
||||
try std.testing.expectEqual(@as(u64, 0), fatToEpoch(0, 0));
|
||||
try std.testing.expectEqual(@as(u16, 0), epochToFatDateTime(0).date);
|
||||
}
|
||||
|
||||
test "on-disk struct sizes match the specification" {
|
||||
try std.testing.expectEqual(@as(usize, 36), @sizeOf(BiosParameterBlock));
|
||||
try std.testing.expectEqual(@as(usize, 26), @sizeOf(ExtendedBootRecord16));
|
||||
try std.testing.expectEqual(@as(usize, 54), @sizeOf(ExtendedBootRecord32));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(DirectoryEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(LongNameEntry));
|
||||
try std.testing.expectEqual(@as(usize, 512), @sizeOf(FileSystemInformation));
|
||||
}
|
||||
|
||||
test "directory entry cluster split/join" {
|
||||
var entry = std.mem.zeroes(DirectoryEntry);
|
||||
entry.setFirstCluster(0x01234567);
|
||||
try std.testing.expectEqual(@as(u16, 0x4567), entry.first_cluster_low);
|
||||
try std.testing.expectEqual(@as(u16, 0x0123), entry.first_cluster_high);
|
||||
try std.testing.expectEqual(@as(u32, 0x01234567), entry.firstCluster());
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//! /system/services/fdt — the devicetree discovery service: the ARM twin of the
|
||||
//! acpi service (docs/m19-m20-plan.md decision 7). **Placeholder: not
|
||||
//! acpi service (docs/discovery.md — firmware neutrality). **Placeholder: not
|
||||
//! implemented.** It exists so the build's `-Ddiscovery` option has both of its
|
||||
//! values from day one; the implementation lands with the Raspberry Pi
|
||||
//! bring-up (docs/arm.md).
|
||||
@@ -15,7 +15,7 @@
|
||||
//! resident under the manager's supervision (hello, restart, the usual
|
||||
//! contract).
|
||||
//!
|
||||
//! Known prerequisite recorded in the plan: `DeviceDescriptor`'s 8-byte `hid`
|
||||
//! Known prerequisite recorded in docs/discovery.md: `DeviceDescriptor`'s 8-byte `hid`
|
||||
//! cannot hold an FDT `compatible` string ("brcm,bcm2835-aux-uart") — identity
|
||||
//! widens before this file grows a body.
|
||||
|
||||
|
||||
+137
-12
@@ -6,18 +6,35 @@
|
||||
//!
|
||||
//! It proves the C-convention heap works, then — as PID 1 — acts as the system's
|
||||
//! **service supervisor**: it spawns the user-space services danos brings up at boot
|
||||
//! (the VFS server, the device manager), and settles into a heartbeat so it stays
|
||||
//! alive as the root of user space. Drivers are *not* its job: the device manager
|
||||
//! discovers the hardware and spawns those. This is the service half of the
|
||||
//! service/driver spawn split (docs/driver-model.md).
|
||||
//! (the VFS server, the device manager), and settles into an event loop as the root
|
||||
//! of user space. Drivers are *not* its job: the device manager discovers the
|
||||
//! hardware and spawns those. This is the service half of the service/driver spawn
|
||||
//! split (docs/driver-model.md).
|
||||
//!
|
||||
//! M21: init also owns **orderly shutdown**. It supervises its children (keeping
|
||||
//! their ids and an exit endpoint), subscribes to the power service, and on a
|
||||
//! power-button event runs the stop sequence over its children in reverse order
|
||||
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
|
||||
//! composing into a clean poweroff.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const power = runtime.power_protocol;
|
||||
|
||||
/// Where the kernel boot log is persisted on the USB FAT volume — an 8.3 name at
|
||||
/// the mount root (see system/services/log-flush). init writes it at shutdown;
|
||||
/// the log-flush one-shot writes it once at boot.
|
||||
const log_path = "/mnt/usb/DANOS.LOG";
|
||||
|
||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager" };
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat" };
|
||||
|
||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var child_count: usize = 0;
|
||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
@@ -28,25 +45,133 @@ pub fn main() void {
|
||||
// the extern malloc/free symbols; Zig code uses this allocator.)
|
||||
const gpa = runtime.allocator();
|
||||
if (gpa.alloc(u8, 64)) |buffer| {
|
||||
const message = "init: heap ok\n";
|
||||
const message = "/system/services/init: heap ok\n";
|
||||
@memcpy(buffer[0..message.len], message);
|
||||
_ = runtime.system.write(buffer[0..message.len]);
|
||||
gpa.free(buffer);
|
||||
} else |_| {}
|
||||
|
||||
// Bring up the boot services. Best-effort and silent: each service announces its
|
||||
// own readiness (`vfs: ready`, ...), and in an isolation test that runs init with
|
||||
// no initial-ramdisk the spawns simply no-op rather than deranging the heartbeat.
|
||||
// One endpoint carries everything init waits on: children's exit
|
||||
// notifications (they are spawned supervised against it), init's own
|
||||
// signals, and power events it subscribes to. All arrive in the loop below.
|
||||
supervision_endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("/system/services/init: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.process.bindSignals(supervision_endpoint);
|
||||
|
||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||
// Best-effort and silent: each service announces its own readiness, and in
|
||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||
for (boot_services) |service| {
|
||||
_ = runtime.system.spawn(service);
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
||||
children[child_count] = id;
|
||||
child_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
||||
// volume (/mnt/usb/DANOS.LOG) so it can be read on another machine — the only
|
||||
// way to see it on a headless/real board with no host capturing serial. Fire
|
||||
// and forget: it polls for the mount itself, and is deliberately NOT one of
|
||||
// init's supervised children (a transient one-shot must not be stopped-and-
|
||||
// waited-for during shutdown).
|
||||
_ = runtime.system.spawn("log-flush");
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
// init starts). Best-effort — without it, a `terminate` signal still
|
||||
// triggers the same shutdown path.
|
||||
subscribePower();
|
||||
|
||||
// A re-arming timer drives the liveness heartbeat: proof PID 1 is alive
|
||||
// (the init test's marker) while the loop stays free to receive signals,
|
||||
// power events, and children's exit notifications.
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
|
||||
var receive: [power.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
_ = runtime.system.write("init: heartbeat\n");
|
||||
runtime.system.sleep(1000);
|
||||
const got = runtime.ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
|
||||
if (runtime.process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isTimer()) {
|
||||
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
continue;
|
||||
}
|
||||
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power.Operation.event)) {
|
||||
// A power event (the only buffered messages init receives).
|
||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||
continue;
|
||||
}
|
||||
// Child-exit notifications and anything else: keep waiting.
|
||||
if (got.isNotification()) continue;
|
||||
}
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
fn subscribePower() void {
|
||||
var handle: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (handle == null and tries < 200) : (tries += 1) {
|
||||
handle = runtime.ipc.lookup(.power);
|
||||
if (handle == null) runtime.system.sleep(20);
|
||||
}
|
||||
// A missing power service is not fatal — init proceeds to its heartbeat and
|
||||
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
|
||||
// test's heartbeat marker is the next line written.
|
||||
const h = handle orelse return;
|
||||
const request = power.Subscribe{};
|
||||
var reply: [power.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
}
|
||||
|
||||
/// Copy the whole kernel log to /mnt/usb/DANOS.LOG (the same file log-flush
|
||||
/// writes at boot), so a poweroff captures the fullest log. Best-effort: if the
|
||||
/// USB volume is not mounted, the open fails and it does nothing. Must run while
|
||||
/// the storage services are still alive (see shutDown).
|
||||
fn flushKernelLog() void {
|
||||
// Truncate on open so this fuller flush replaces the boot-time one cleanly.
|
||||
var file = runtime.fs.open(log_path, .{ .create = true, .truncate = true }) orelse return; // no USB volume
|
||||
defer file.close();
|
||||
var chunk: [4096]u8 = undefined;
|
||||
var offset: usize = 0;
|
||||
while (true) {
|
||||
const got = runtime.system.klogRead(offset, &chunk);
|
||||
if (got == 0) break; // reached the end of the accumulated log
|
||||
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||
offset += got;
|
||||
}
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "/system/services/init: flushed log to {s} ({d} bytes)\n", .{ log_path, offset }) catch "");
|
||||
}
|
||||
|
||||
/// The stop sequence: persist the log while storage is still up, then terminate
|
||||
/// each child in reverse spawn order (vfs last — other services may flush through
|
||||
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
||||
/// power service to enter S5.
|
||||
fn shutDown() void {
|
||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
||||
// reverse-order stop loop below kills the fat server (children[3]) first, so
|
||||
// /mnt/usb must be written while it is still mounted.
|
||||
flushKernelLog();
|
||||
var i = child_count;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
||||
}
|
||||
if (runtime.ipc.lookup(.power)) |h| {
|
||||
const request = power.Shutdown{};
|
||||
var reply: [power.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
// If S5 did not take, init has nothing left to do but idle.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
|
||||
@@ -115,14 +115,14 @@ fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
|
||||
pub fn main() void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = system.write("input: no endpoint\n");
|
||||
_ = system.write("/system/services/input: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
if (!ipc.register(.input, endpoint)) {
|
||||
_ = system.write("input: register failed\n");
|
||||
_ = system.write("/system/services/input: register failed\n");
|
||||
return;
|
||||
}
|
||||
_ = system.write("input: ready\n");
|
||||
_ = system.write("/system/services/input: ready\n");
|
||||
|
||||
var reply_buffer: [protocol.reply_size]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
//! system/services/log-flush — a one-shot that copies the kernel's in-memory
|
||||
//! diagnostic log to a file on the mounted USB FAT volume, so the boot log
|
||||
//! survives to be read on another machine. On a headless or real board there is
|
||||
//! no host capturing serial, so without this the log is lost at power-off; this
|
||||
//! is the on-disk equivalent of QEMU's `-serial file:`.
|
||||
//!
|
||||
//! It reads the whole kernel log back through `klog_read` (the RAM sink in
|
||||
//! system/kernel/log.zig) and writes it to /mnt/usb/DANOS.LOG. The name is 8.3
|
||||
//! (FAT short-name rule: base <= 8, extension <= 3) and lives at the mount root
|
||||
//! (there is no mkdir on the FAT path yet). init spawns this once the boot
|
||||
//! services are up; init itself repeats the flush at shutdown for a fuller log.
|
||||
//!
|
||||
//! If no USB volume is mounted — no stick, or the initial-ramdisk sweep that
|
||||
//! spawns every bundled binary bare with no VFS — it waits briefly, then exits
|
||||
//! silently, deranging no other test's output.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
const log_path = "/mnt/usb/DANOS.LOG";
|
||||
|
||||
/// Copy the whole kernel log to the open file, looping klog_read -> write until
|
||||
/// the log is exhausted. Returns the number of bytes written.
|
||||
fn drainKernelLog(file: *fs.File) usize {
|
||||
var chunk: [4096]u8 = undefined;
|
||||
var offset: usize = 0;
|
||||
while (true) {
|
||||
const got = runtime.system.klogRead(offset, &chunk);
|
||||
if (got == 0) break; // reached the end of the accumulated log
|
||||
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||
offset += got;
|
||||
}
|
||||
return offset;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Wait for the fat server to mount /mnt/usb (it must bring up the whole USB
|
||||
// storage chain first, so it races us at boot). Bounded: if the mount never
|
||||
// appears — no volume, or the no-VFS ramdisk sweep — give up silently.
|
||||
var ready = false;
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1400) : (tries += 1) {
|
||||
if (fs.openDirectory("/mnt/usb")) |directory| {
|
||||
var dir = directory;
|
||||
dir.close();
|
||||
ready = true;
|
||||
break;
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
}
|
||||
if (!ready) return; // /mnt/usb never became available — nothing to persist to
|
||||
|
||||
// Truncate on open: each flush replaces the file, so a shorter log on a later
|
||||
// boot of the same stick leaves no stale tail from a previous, longer one.
|
||||
var file = fs.open(log_path, .{ .create = true, .truncate = true }) orelse return;
|
||||
const written = drainKernelLog(&file);
|
||||
file.close();
|
||||
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "log-flush: wrote {d} bytes to {s}\n", .{ written, log_path }) catch return);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
//! The power protocol (docs/power.md): system power's domain-named surface,
|
||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
||||
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
|
||||
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||
|
||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
pub const Operation = enum(u8) {
|
||||
/// Subscribe to power events: the subscriber's endpoint rides as the
|
||||
/// call's capability (the input/device-manager pattern); events arrive on
|
||||
/// it as buffered messages carrying an `EventMessage`.
|
||||
subscribe = 1,
|
||||
/// Orderly shutdown's last step: enter S5. Accepted only from PID 1
|
||||
/// (init) — the process that has already run the stop sequence over
|
||||
/// everything else.
|
||||
shutdown = 2,
|
||||
/// The published event payload (never sent *to* the service).
|
||||
event = 3,
|
||||
};
|
||||
|
||||
/// What happened. The vocabulary is hardware-neutral: a lid is a lid whether
|
||||
/// ACPI or a PSCI mailbox reported it.
|
||||
pub const Event = enum(u8) {
|
||||
power_button = 1,
|
||||
lid = 2,
|
||||
ac = 3,
|
||||
battery = 4,
|
||||
/// A device notification that maps to none of the named events — the
|
||||
/// `code` and `hid` fields say which device and what code.
|
||||
notify = 5,
|
||||
};
|
||||
|
||||
pub const Subscribe = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Shutdown = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.shutdown),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
/// A published event, as the buffered-message payload subscribers receive.
|
||||
pub const EventMessage = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.event),
|
||||
/// An Event value.
|
||||
event: u8,
|
||||
reserved0: u16 = 0,
|
||||
/// The device notification code (Notify's second argument), or 0.
|
||||
code: u32 = 0,
|
||||
/// The notifying device's hardware id (EISA-decoded), or all zero.
|
||||
hid: [8]u8 = .{0} ** 8,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// Upper bound on any message in this protocol — sizes endpoint buffers.
|
||||
pub const message_maximum = 64;
|
||||
@@ -0,0 +1,39 @@
|
||||
//! Pure path utilities for the VFS mount router — no IPC, no state, so they are
|
||||
//! host-testable in isolation. The router uses these to decide whether an opened
|
||||
//! path lies under a mount point and, if so, what it looks like relative to that
|
||||
//! mount.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by a
|
||||
/// path separator — return the path relative to the mount ("/" for an exact
|
||||
/// match, otherwise the tail beginning with '/'). Returns null when `path` is not
|
||||
/// under the mount, so a prefix like "/mnt/usb" never captures "/mnt/usbextra".
|
||||
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||
if (path.len < mount_prefix.len) return null;
|
||||
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||
if (path.len == mount_prefix.len) return "/";
|
||||
if (path[mount_prefix.len] != '/') return null;
|
||||
return path[mount_prefix.len..];
|
||||
}
|
||||
|
||||
/// Whether `path` is absolute (rooted at '/'). Bare names — what the flat ramfs
|
||||
/// uses — are relative and never route through a mount.
|
||||
pub fn isAbsolute(path: []const u8) bool {
|
||||
return path.len > 0 and path[0] == '/';
|
||||
}
|
||||
|
||||
test "underMount matches only at path boundaries" {
|
||||
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
||||
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
||||
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null); // not a boundary
|
||||
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null); // shorter than the prefix
|
||||
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("greeting", "/mnt/usb") == null); // a bare name
|
||||
}
|
||||
|
||||
test "isAbsolute distinguishes paths from bare names" {
|
||||
try std.testing.expect(isAbsolute("/mnt/usb"));
|
||||
try std.testing.expect(!isAbsolute("greeting"));
|
||||
try std.testing.expect(!isAbsolute(""));
|
||||
}
|
||||
@@ -4,12 +4,11 @@
|
||||
//! header followed by an inline payload (read bytes, or a FileStatus). Everything fits
|
||||
//! in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
|
||||
//!
|
||||
//! This is a danos-native contract, so it uses danos names throughout — the POSIX
|
||||
//! spellings (`stat`, `O_CREAT`, ...) live only in the POSIX layer
|
||||
//! (library/posix/unistd.zig), which translates to these.
|
||||
//! This is a danos-native contract, so it uses danos names throughout. The client
|
||||
//! side is `runtime.fs` (library/runtime/fs.zig), which programs use directly.
|
||||
//!
|
||||
//! This is user-space only — the kernel knows nothing of files or paths; it only moves the bytes.
|
||||
//! Shared by library/posix/unistd.zig (client) and system/services/vfs/vfs.zig (server).
|
||||
//! Shared by library/runtime/fs.zig (client) and system/services/vfs/vfs.zig (server).
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
open, // open(path) -> node id
|
||||
@@ -17,8 +16,43 @@ pub const Operation = enum(u32) {
|
||||
read, // read(node, offset, len) -> bytes
|
||||
write, // write(node, offset, bytes) -> count
|
||||
status, // status(node) -> FileStatus
|
||||
// Appended for the mount router (M5). Values stay stable, so existing clients
|
||||
// and the flat-ramfs tests are unaffected.
|
||||
readdir, // readdir(dir_node, cursor=offset) -> one DirectoryEntry (len==0 => EOF)
|
||||
mount, // mount(prefix payload, capability = backend endpoint)
|
||||
unmount, // unmount(prefix payload)
|
||||
// Appended for filesystem mutation (Phase 2). Path-based (the path is the
|
||||
// payload); a mounted backend handles them, the flat ramfs refuses them.
|
||||
mkdir, // mkdir(path payload) -> status
|
||||
unlink, // unlink(path payload) -> status
|
||||
// rename: the payload is the old path, a single 0x00 separator, then the new
|
||||
// path. Same-directory rename only (the router requires both under one mount).
|
||||
rename, // rename(old\0new payload) -> status
|
||||
};
|
||||
|
||||
/// The type of a filesystem node, aligned to the FSH file-type table
|
||||
/// (docs/danos-file-system-hierarchy-FSH.md). Fills `FileStatus.kind` and
|
||||
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
|
||||
pub const NodeKind = enum(u32) {
|
||||
regular = 0,
|
||||
directory = 1,
|
||||
character_device = 2,
|
||||
block_device = 3,
|
||||
symbolic_link = 4,
|
||||
fifo = 5,
|
||||
socket = 6,
|
||||
};
|
||||
|
||||
/// One directory entry, returned by `readdir`: a fixed header followed inline in
|
||||
/// the reply payload by `name_len` bytes of name. A zero-length reply is EOF.
|
||||
pub const DirectoryEntry = extern struct {
|
||||
kind: u32, // a NodeKind
|
||||
name_len: u32,
|
||||
size: u64,
|
||||
};
|
||||
|
||||
pub const directory_entry_size: usize = @sizeOf(DirectoryEntry);
|
||||
|
||||
/// Request header. `node` is the server-side open-file id (from a prior open);
|
||||
/// for `open` the path is the payload and `len` is its length. `offset`/`len`
|
||||
/// carry the read/write position and count.
|
||||
@@ -47,6 +81,9 @@ pub const FileStatus = extern struct {
|
||||
size: u64,
|
||||
kind: u32,
|
||||
_padding: u32 = 0,
|
||||
/// Modification time — Unix epoch seconds, UTC. 0 if the backend has none (the
|
||||
/// flat ramfs). Filled from the FAT directory entry's write date/time.
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
pub const message_maximum: usize = 256;
|
||||
@@ -55,5 +92,23 @@ pub const reply_size: usize = @sizeOf(Reply);
|
||||
/// Largest inline payload that still fits one IPC message alongside a header.
|
||||
pub const maximum_payload: usize = message_maximum - request_size;
|
||||
|
||||
/// Open flags (danos-native; the POSIX layer maps `O_CREAT` onto `create`).
|
||||
/// Open flags (danos-native; `runtime.fs.OpenOptions` maps its booleans onto these).
|
||||
pub const create: u32 = 1;
|
||||
/// Open a directory (for readdir) rather than a file. A mounted backend uses
|
||||
/// this to open a directory node; the flat ramfs ignores it.
|
||||
pub const directory: u32 = 2;
|
||||
/// Truncate the file to zero length on open (O_TRUNC): replace its contents rather
|
||||
/// than overwriting in place, so a shorter new file leaves no stale tail. A mounted
|
||||
/// backend frees the old cluster chain; the flat ramfs ignores it.
|
||||
pub const truncate: u32 = 4;
|
||||
|
||||
test "protocol struct sizes and node kinds" {
|
||||
const std = @import("std");
|
||||
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(NodeKind.regular));
|
||||
try std.testing.expectEqual(@as(u32, 1), @intFromEnum(NodeKind.directory));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(DirectoryEntry));
|
||||
// The appended operations keep the original values.
|
||||
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(Operation.open));
|
||||
try std.testing.expectEqual(@as(u32, 4), @intFromEnum(Operation.status));
|
||||
try std.testing.expectEqual(@as(u32, 5), @intFromEnum(Operation.readdir));
|
||||
}
|
||||
|
||||
@@ -1,26 +1,26 @@
|
||||
//! /system/services/vfs/vfs-test — a client that proves the VFS round trip end to end: open a
|
||||
//! file through the `runtime` file API, write to it, seek back, read it, and compare.
|
||||
//! file through the `runtime.fs` file API, write to it, seek back, read it, and compare.
|
||||
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
|
||||
//! on failure it reports what went wrong. Shipped in the initial_ramdisk alongside vfs.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const u = @import("posix").unistd;
|
||||
const payload = "hello-vfs";
|
||||
|
||||
// The "park" role (the vfs-client-death test): open a file, then hold the
|
||||
// handle forever without closing — the kill and the VFS's release-on-death
|
||||
// are the point.
|
||||
if (init.arguments.count > 1) {
|
||||
var fd: i32 = -1;
|
||||
var parked: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (fd < 0 and tries < 200) : (tries += 1) {
|
||||
fd = u.open("parked", u.O_CREAT);
|
||||
if (fd < 0) runtime.system.sleep(20);
|
||||
while (parked == null and tries < 200) : (tries += 1) {
|
||||
parked = fs.open("parked", .{ .create = true });
|
||||
if (parked == null) runtime.system.sleep(20);
|
||||
}
|
||||
if (fd < 0) {
|
||||
if (parked == null) {
|
||||
_ = runtime.system.write("vfstest: park open failed\n");
|
||||
return;
|
||||
}
|
||||
@@ -31,28 +31,28 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
|
||||
// The VFS server may not have registered yet — retry open until it's up.
|
||||
var fd: i32 = -1;
|
||||
var opened: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (fd < 0 and tries < 200) : (tries += 1) {
|
||||
fd = u.open("greeting", u.O_CREAT);
|
||||
if (fd < 0) runtime.system.sleep(20);
|
||||
while (opened == null and tries < 200) : (tries += 1) {
|
||||
opened = fs.open("greeting", .{ .create = true });
|
||||
if (opened == null) runtime.system.sleep(20);
|
||||
}
|
||||
if (fd < 0) {
|
||||
var greeting = opened orelse {
|
||||
_ = runtime.system.write("vfstest: open failed\n");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if (u.write(fd, payload) != @as(isize, payload.len)) {
|
||||
if ((greeting.write(payload) orelse 0) != payload.len) {
|
||||
_ = runtime.system.write("vfstest: write failed\n");
|
||||
return;
|
||||
}
|
||||
_ = u.lseek(fd, 0, u.SEEK_SET);
|
||||
greeting.seekTo(0);
|
||||
|
||||
var buffer: [32]u8 = undefined;
|
||||
const n = u.read(fd, &buffer);
|
||||
u.close(fd);
|
||||
const n = greeting.read(&buffer) orelse 0;
|
||||
greeting.close();
|
||||
|
||||
if (n == @as(isize, payload.len) and std.mem.eql(u8, buffer[0..@intCast(n)], payload)) {
|
||||
if (n == payload.len and std.mem.eql(u8, buffer[0..n], payload)) {
|
||||
while (true) {
|
||||
_ = runtime.system.write("vfstest: ok\n");
|
||||
runtime.system.sleep(1000);
|
||||
|
||||
+245
-19
@@ -3,14 +3,24 @@
|
||||
//! file API marshals open/read/write/stat/close into calls to this server's
|
||||
//! endpoint, published under the well-known `vfs` service id).
|
||||
//!
|
||||
//! For now the namespace is a small in-memory ramfs (opening a name creates it):
|
||||
//! enough to prove the whole path — client file API -> IPC -> server dispatch ->
|
||||
//! reply. Device nodes backed by user-space drivers (/device) layer on top in M10,
|
||||
//! where `open` on a /device name forwards to the owning driver's endpoint.
|
||||
//! Two namespaces meet here (M5):
|
||||
//! - a small in-memory **ramfs** — opening a bare name creates it — enough to
|
||||
//! prove the round trip and to back the existing tests;
|
||||
//! - **mounted filesystems**: a mount table maps an absolute path prefix (e.g.
|
||||
//! `/mnt/usb`) to a backend server's endpoint. An open of a path under a mount
|
||||
//! is *forwarded* to that backend (which speaks this same protocol), and every
|
||||
//! later read/write/status/readdir/close on the resulting handle is relayed to
|
||||
//! it. The VFS is the router; a filesystem (FAT) is the backend.
|
||||
//!
|
||||
//! A path routes through a mount only when it is absolute and lies under a mount
|
||||
//! prefix; bare names always resolve in the flat ramfs — the backward-compat
|
||||
//! contract the `vfs` / `vfs-client-death` tests rely on.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.vfs_protocol;
|
||||
const path = @import("path.zig");
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const Node = struct {
|
||||
used: bool = false,
|
||||
@@ -22,15 +32,29 @@ const Node = struct {
|
||||
|
||||
const OpenFile = struct {
|
||||
used: bool = false,
|
||||
// For a local handle: an index into `nodes`. For a forwarding handle: the
|
||||
// node id the backend returned. (usize == u64 here, so it holds either.)
|
||||
node: usize = 0,
|
||||
// Non-null for a handle that forwards to a mounted backend.
|
||||
backend: ?ipc.Handle = null,
|
||||
// The client (task id — an IPC badge is one) that opened this handle. What
|
||||
// release-on-death sweeps by: a service must never depend on its clients
|
||||
// cleaning up after themselves (docs/process-lifecycle.md).
|
||||
owner: u32 = 0,
|
||||
};
|
||||
|
||||
// One mounted filesystem: an absolute path prefix and the backend endpoint that
|
||||
// serves everything under it.
|
||||
const Mount = struct {
|
||||
used: bool = false,
|
||||
prefix: [64]u8 = undefined,
|
||||
prefix_len: usize = 0,
|
||||
backend: ipc.Handle = 0,
|
||||
};
|
||||
|
||||
var nodes = [_]Node{.{}} ** 8;
|
||||
var opens = [_]OpenFile{.{}} ** 16;
|
||||
var mounts = [_]Mount{.{}} ** 8;
|
||||
|
||||
fn findNode(name: []const u8) ?usize {
|
||||
for (&nodes, 0..) |*n, i| {
|
||||
@@ -57,6 +81,26 @@ fn openAt(id: u64) ?*OpenFile {
|
||||
return if (o.used) o else null;
|
||||
}
|
||||
|
||||
/// The mount whose prefix most specifically contains `name`, and the path
|
||||
/// relative to it. Only absolute paths route; bare names never match.
|
||||
const MountMatch = struct { backend: ipc.Handle, relative: []const u8 };
|
||||
fn longestMount(name: []const u8) ?MountMatch {
|
||||
if (!path.isAbsolute(name)) return null;
|
||||
var best: ?MountMatch = null;
|
||||
var best_len: usize = 0;
|
||||
for (&mounts) |*m| {
|
||||
if (!m.used) continue;
|
||||
const prefix = m.prefix[0..m.prefix_len];
|
||||
if (path.underMount(name, prefix)) |relative| {
|
||||
if (best == null or prefix.len >= best_len) {
|
||||
best_len = prefix.len;
|
||||
best = .{ .backend = m.backend, .relative = relative };
|
||||
}
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/// Serialise a reply header + payload into `out`; returns the total length.
|
||||
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
|
||||
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
|
||||
@@ -76,34 +120,168 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// --- mount routing ----------------------------------------------------------
|
||||
|
||||
/// Forward an open under a mount to its backend and, on success, allocate a local
|
||||
/// forwarding handle that remembers the backend's node id.
|
||||
fn forwardOpen(out: []u8, backend: ipc.Handle, relative: []const u8, flags: u32, sender: u32) usize {
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = flags };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const rel = relative[0..@min(relative.len, protocol.maximum_payload)];
|
||||
@memcpy(message[protocol.request_size..][0..rel.len], rel);
|
||||
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out);
|
||||
if (n < protocol.reply_size) return fail(out);
|
||||
const backend_reply = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (backend_reply.status != 0) return writeReply(out, .{ .status = backend_reply.status }, &.{});
|
||||
|
||||
for (&opens, 0..) |*o, i| {
|
||||
if (!o.used) {
|
||||
o.* = .{ .used = true, .node = @intCast(backend_reply.node), .backend = backend, .owner = sender };
|
||||
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
}
|
||||
|
||||
/// Relay a read/write/status/readdir/close on a forwarding handle to the backend
|
||||
/// (the node already rewritten to the backend's id) and copy its reply out.
|
||||
fn forwardRequest(out: []u8, backend: ipc.Handle, request: protocol.Request, payload: []const u8) usize {
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const plen = @min(payload.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..plen], payload[0..plen]);
|
||||
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0 .. protocol.request_size + plen], &reply) catch return fail(out);
|
||||
const copy = @min(n, out.len);
|
||||
@memcpy(out[0..copy], reply[0..copy]);
|
||||
return copy;
|
||||
}
|
||||
|
||||
/// Forward a path-based operation (mkdir, unlink) under a mount to its backend and
|
||||
/// relay the reply. No handle is created — these operate by path and return only a
|
||||
/// status.
|
||||
fn forwardPath(out: []u8, backend: ipc.Handle, operation: protocol.Operation, relative: []const u8) usize {
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const rel = relative[0..@min(relative.len, protocol.maximum_payload)];
|
||||
@memcpy(message[protocol.request_size..][0..rel.len], rel);
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out);
|
||||
const copy = @min(n, out.len);
|
||||
@memcpy(out[0..copy], reply[0..copy]);
|
||||
return copy;
|
||||
}
|
||||
|
||||
/// Forward a rename to its backend: the payload is the mount-relative old path, a
|
||||
/// 0x00 separator, then the mount-relative new path. Relays the backend's reply.
|
||||
fn forwardRename(out: []u8, backend: ipc.Handle, old_relative: []const u8, new_relative: []const u8) usize {
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
if (protocol.request_size + total > message.len) return fail(out);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
var p = protocol.request_size;
|
||||
@memcpy(message[p..][0..old_relative.len], old_relative);
|
||||
p += old_relative.len;
|
||||
message[p] = 0;
|
||||
p += 1;
|
||||
@memcpy(message[p..][0..new_relative.len], new_relative);
|
||||
p += new_relative.len;
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0..p], &reply) catch return fail(out);
|
||||
const copy = @min(n, out.len);
|
||||
@memcpy(out[0..copy], reply[0..copy]);
|
||||
return copy;
|
||||
}
|
||||
|
||||
/// Best-effort close of a backend node (used when a dead client's forwarding
|
||||
/// handles are swept — the backend must not leak the vfs's opens).
|
||||
fn forwardClose(backend: ipc.Handle, backend_node: u64) void {
|
||||
const request = protocol.Request{ .operation = .close, .node = backend_node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.call(backend, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
|
||||
fn doMount(out: []u8, prefix: []const u8, backend: ipc.Handle) usize {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) {
|
||||
m.backend = backend;
|
||||
writeLine("/system/services/vfs: remounted {s}\n", .{prefix});
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
}
|
||||
}
|
||||
for (&mounts) |*m| {
|
||||
if (!m.used) {
|
||||
const l = @min(prefix.len, m.prefix.len);
|
||||
m.used = true;
|
||||
@memcpy(m.prefix[0..l], prefix[0..l]);
|
||||
m.prefix_len = l;
|
||||
m.backend = backend;
|
||||
writeLine("/system/services/vfs: mounted {s}\n", .{prefix[0..l]});
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
}
|
||||
|
||||
fn doUnmount(out: []u8, prefix: []const u8) usize {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) {
|
||||
m.used = false;
|
||||
writeLine("/system/services/vfs: unmounted {s}\n", .{prefix});
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
}
|
||||
|
||||
/// Release every open handle `client` held — called on that client's published
|
||||
/// exit event. The nodes (the files) stay: ramfs contents outlive their writers,
|
||||
/// only the dead client's handles go.
|
||||
/// exit event. Forwarding handles also tell their backend to release; local
|
||||
/// nodes (the ramfs files) stay, since ramfs contents outlive their writers.
|
||||
fn releaseClientHandles(client: u32) void {
|
||||
var released: u32 = 0;
|
||||
for (&opens) |*o| {
|
||||
if (o.used and o.owner == client) {
|
||||
if (o.backend) |backend| forwardClose(backend, o.node);
|
||||
o.used = false;
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) writeLine("vfs: released {d} handle(s) for dead client {d}\n", .{ released, client });
|
||||
if (released != 0) writeLine("/system/services/vfs: released {d} handle(s) for dead client {d}\n", .{ released, client });
|
||||
}
|
||||
|
||||
/// Handle one request from `sender`; write the reply into `out`, return its length.
|
||||
fn handle(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = capability;
|
||||
fn handle(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
|
||||
switch (request.operation) {
|
||||
.mount => {
|
||||
const prefix = payload[0..@min(payload.len, request.len)];
|
||||
const backend = capability orelse return fail(out);
|
||||
return doMount(out, prefix, backend);
|
||||
},
|
||||
.unmount => {
|
||||
const prefix = payload[0..@min(payload.len, request.len)];
|
||||
return doUnmount(out, prefix);
|
||||
},
|
||||
.open => {
|
||||
const name = payload[0..@min(payload.len, request.len)];
|
||||
if (longestMount(name)) |m| return forwardOpen(out, m.backend, m.relative, request.flags, sender);
|
||||
// An absolute path with no matching mount is simply not found — only
|
||||
// bare names live in the flat ramfs. (Else /mnt/usb would be silently
|
||||
// created as a flat file when its filesystem is not yet mounted.)
|
||||
if (path.isAbsolute(name)) return fail(out);
|
||||
const ni = findNode(name) orelse createNode(name) orelse return fail(out);
|
||||
for (&opens, 0..) |*o, i| {
|
||||
if (!o.used) {
|
||||
o.* = .{ .used = true, .node = ni, .owner = sender };
|
||||
o.* = .{ .used = true, .node = ni, .backend = null, .owner = sender };
|
||||
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
||||
}
|
||||
}
|
||||
@@ -111,7 +289,12 @@ fn handle(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.
|
||||
},
|
||||
.read => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const nd = &nodes[of.node];
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
const nd = &nodes[@intCast(of.node)];
|
||||
const off: usize = @intCast(request.offset);
|
||||
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
|
||||
const n = @min(@min(nd.size - off, request.len), protocol.maximum_payload);
|
||||
@@ -119,7 +302,12 @@ fn handle(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.
|
||||
},
|
||||
.write => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const nd = &nodes[of.node];
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
const nd = &nodes[@intCast(of.node)];
|
||||
const off: usize = @intCast(request.offset);
|
||||
if (off > nd.data.len) return fail(out);
|
||||
const n = @min(@min(payload.len, request.len), nd.data.len - off);
|
||||
@@ -129,31 +317,69 @@ fn handle(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.
|
||||
},
|
||||
.status => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const st = protocol.FileStatus{ .size = nodes[of.node].size, .kind = 0 };
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
const st = protocol.FileStatus{ .size = nodes[@intCast(of.node)].size, .kind = @intFromEnum(protocol.NodeKind.regular) };
|
||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&st));
|
||||
},
|
||||
.readdir => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
// The flat ramfs has no directories: report EOF.
|
||||
return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
|
||||
},
|
||||
.close => {
|
||||
if (request.node < opens.len) opens[@intCast(request.node)].used = false;
|
||||
const of = openAt(request.node);
|
||||
if (of) |o| {
|
||||
if (o.backend) |backend| forwardClose(backend, o.node);
|
||||
o.used = false;
|
||||
}
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.mkdir, .unlink => {
|
||||
const name = payload[0..@min(payload.len, request.len)];
|
||||
if (longestMount(name)) |m| return forwardPath(out, m.backend, request.operation, m.relative);
|
||||
// Only a mounted backend has real directories; the flat ramfs cannot
|
||||
// create or remove them (and a bare-name path is not a mount target).
|
||||
return fail(out);
|
||||
},
|
||||
.rename => {
|
||||
const both = payload[0..@min(payload.len, request.len)];
|
||||
const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out);
|
||||
const old_path = both[0..sep];
|
||||
const new_path = both[sep + 1 ..];
|
||||
const mo = longestMount(old_path) orelse return fail(out);
|
||||
const mn = longestMount(new_path) orelse return fail(out);
|
||||
// Both paths must live under the same mount — cross-filesystem rename is
|
||||
// not supported.
|
||||
if (mo.backend != mn.backend) return fail(out);
|
||||
return forwardRename(out, mo.backend, mo.relative, mn.relative);
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Startup, under the harness: subscribe to the published exit events — when a
|
||||
/// client dies holding open handles, the exit notification is how the VFS learns
|
||||
/// to release them (docs/process-lifecycle.md).
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
if (!runtime.process.subscribeExits(endpoint)) {
|
||||
_ = runtime.system.write("vfs: exit subscription failed\n");
|
||||
_ = runtime.system.write("/system/services/vfs: exit subscription failed\n");
|
||||
}
|
||||
_ = runtime.system.write("vfs: ready\n");
|
||||
_ = runtime.system.write("/system/services/vfs: ready\n");
|
||||
return true;
|
||||
}
|
||||
|
||||
/// A non-signal notification: the only kind the VFS subscribes to is exit events.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & runtime.ipc.notify_exit_bit != 0) {
|
||||
releaseClientHandles(@intCast(badge & ~(runtime.ipc.notify_badge_bit | runtime.ipc.notify_exit_bit)));
|
||||
if (badge & ipc.notify_exit_bit != 0) {
|
||||
releaseClientHandles(@intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit)));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+200
-64
@@ -18,11 +18,14 @@ Usage:
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
@@ -55,20 +58,23 @@ ARCHES = {
|
||||
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Intel)
|
||||
],
|
||||
# zig-out is a FHS-shaped image and the boot volume; the harness copies the
|
||||
# boot-critical files from their FHS paths into a fresh ESP with the same
|
||||
# layout. (dest in ESP, source path under zig-out) — identical here.
|
||||
"efi_app": ("EFI/BOOT/BOOTX64.efi", "EFI/BOOT/BOOTX64.efi"),
|
||||
"kernel": ("system/kernel", "system/kernel"),
|
||||
# The init user program and the initial-ramdisk (VFS server + drivers).
|
||||
"extra": [("system/services/init", "system/services/init"),
|
||||
("boot/initial-ramdisk.img", "boot/initial-ramdisk.img")],
|
||||
# zig-out is itself the FHS-shaped boot volume (docs/efi.md): the build
|
||||
# installs BOOTX64.efi, the kernel, init, and the initial-ramdisk at their
|
||||
# boot paths. The harness presents zig-out to the guest directly — exactly
|
||||
# as `zig build run-x86-64` does — so there is no separate ESP to assemble.
|
||||
# Built as a function so we can splice in per-run paths.
|
||||
"qemu_args": lambda a, esp, vars_fd, serial: [
|
||||
"qemu_args": lambda a, boot_volume, vars_fd, serial: [
|
||||
"-machine", "q35", "-m", "128M",
|
||||
"-drive", f"if=pflash,format=raw,readonly=on,file={a['ovmf_code']}",
|
||||
"-drive", f"if=pflash,format=raw,file={vars_fd}",
|
||||
"-drive", f"format=raw,file=fat:rw:{esp}",
|
||||
# Boot off a FAT USB device: the boot volume is a mass-storage device on
|
||||
# the xHCI bus (usb-kbd/usb-mouse ride the same controller). `boot_volume`
|
||||
# is the FAT image the build produces. bootindex=0 steers OVMF to it.
|
||||
"-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0",
|
||||
"-drive", f"if=none,id=bootusb,format=raw,file={boot_volume}",
|
||||
"-device", "usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net", "none",
|
||||
"-vga", "none", "-device", "VGA,edid=on,xres=1280,yres=720",
|
||||
"-display", "none",
|
||||
@@ -84,7 +90,10 @@ ARCHES = {
|
||||
# `expect`: a regex that must appear in serial output => pass.
|
||||
# `fail`: optional regex whose appearance => immediate fail.
|
||||
CASES = [
|
||||
# smoke also proves the QMP channel: the harmless query must be delivered
|
||||
# (handshake + command) before the case may pass — see run_case.
|
||||
{"name": "smoke",
|
||||
"qmp_after": {"delay": 2, "command": "query-status"},
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
{"name": "discovery",
|
||||
@@ -99,6 +108,11 @@ CASES = [
|
||||
{"name": "clock",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Wall-clock: the CMOS RTC read at boot gives a plausible current epoch (the
|
||||
# foundation for filesystem mtime).
|
||||
{"name": "wall-clock",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
{"name": "vmm",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -176,6 +190,15 @@ CASES = [
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# TSC clocksource + cross-core warp check (real Intel/AMD / KVM path). TCG won't
|
||||
# advertise an invariant TSC, so the kernel forces the TSC clocksource on for this
|
||||
# case (gated in kernel.zig) and runs the per-AP warp check across the 4 cores;
|
||||
# their TSCs are synchronized, so it stays on the TSC (no HPET fallback). The rest
|
||||
# of the suite exercises the HPET fallback instead. See docs/timers.md.
|
||||
{"name": "tsc-sync",
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
{"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"},
|
||||
{"name": "fault-pf", "expect": r"page fault \(vector 14\)"},
|
||||
{"name": "fault-df", "expect": r"double fault \(vector 8\)"},
|
||||
@@ -264,9 +287,8 @@ CASES = [
|
||||
{"name": "usb-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
# The xHCI bus + usb-kbd/usb-mouse come from the default boot config now
|
||||
# (every case boots off a usb-storage device on that bus).
|
||||
"expect": r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
@@ -274,16 +296,79 @@ CASES = [
|
||||
r"device-manager: restarting usb-xhci-bus[\s\S]*"
|
||||
r"device-manager: child added",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# USB HID end to end: boot the full tree, enumerate the xHCI, and let the
|
||||
# manager spawn the USB keyboard driver, which opens its device over the
|
||||
# transfer protocol, asks for boot protocol, subscribes to its interrupt
|
||||
# endpoint, and comes up — proof the class-driver <-> controller path works.
|
||||
{"name": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args).
|
||||
"expect": r"(?=[\s\S]*usb-hid/keyboard: ok)(?=[\s\S]*usb-hid/mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# USB mass storage end to end: the boot usb-storage device (the FAT32 image,
|
||||
# which has a real 0x55AA boot sector) is enough — the manager spawns
|
||||
# usb-storage, which opens the device, runs the Bulk-Only / SCSI bring-up,
|
||||
# reads its capacity, and reads block 0 (the 0x55AA boot sig). Proof of the
|
||||
# bulk transfer path + BOT + SCSI end to end.
|
||||
{"name": "usb-storage",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
||||
# FAT32 image) into the VFS at /mnt/usb. A fat-test client then lists and reads
|
||||
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
||||
# VFS routing -> file read.
|
||||
{"name": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat: mounted /mnt/usb[\s\S]*fat-test: ok",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
||||
{"name": "fat-mutations",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat-test: mutations ok",
|
||||
"fail": r"fat-test: mutations FAILED|fat-test: mkdir .* failed|DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2c: rename through the mount — fat-test renames the file it created
|
||||
# before removing it, and confirms the old name is gone.
|
||||
{"name": "fat-rename",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat-test: rename ok",
|
||||
"fail": r"fat-test: mutations FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2d: filesystem timestamps — a freshly-created file's mtime is a real
|
||||
# current wall-clock time (stamped from the RTC), read back through stat.
|
||||
{"name": "fat-mtime",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat-test: mtime ok",
|
||||
"fail": r"fat-test: mutations FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# Boot-from-USB smoke: the whole system now boots off the FAT32 image on a
|
||||
# usb-storage device (OVMF -> \EFI\BOOT\BOOTX64.efi -> kernel), so the kernel
|
||||
# reaching its PASS marker at all proves the USB boot path end to end. Reuses
|
||||
# the smoke kernel build; the value is the explicit, named regression guard.
|
||||
{"name": "usb-boot",
|
||||
"build_case": "smoke",
|
||||
"qmp_after": {"delay": 2, "command": "query-status"},
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.1: the ring-3 AML parse (the acpi service maps the blobs and parses
|
||||
# them) finds exactly the Device count the kernel's own parse produced.
|
||||
{"name": "acpi-parse",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"acpi-parse: ok",
|
||||
"fail": r"acpi-parse: mismatch|DANOS-TEST-RESULT: FAIL"},
|
||||
"fail": r"acpi-parse: too few|DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
|
||||
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
|
||||
# keyboard, proving discovery runs entirely in ring 3 (docs/m19-m20-plan.md).
|
||||
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
|
||||
{"name": "acpi-ps2",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
@@ -291,15 +376,48 @@ CASES = [
|
||||
r"device-manager: spawned ps2-bus[\s\S]*"
|
||||
r"ps2-bus: keyboard driver attached",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
||||
# service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed
|
||||
# event; the service's SCI handler must log the press (docs/acpi.md).
|
||||
{"name": "power-button",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"qmp_after": {"delay": 4, "command": "system_powerdown"},
|
||||
"expect": r"power: button pressed",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M21.3 capstone: orderly shutdown. Boot init (the full tree comes up);
|
||||
# ~5s in, QMP system_powerdown raises the power button; the acpi service
|
||||
# publishes it, init stops its children then requests S5, and QEMU exits.
|
||||
# The ordered regex proves button -> shutting-down -> entering-S5; the case
|
||||
# passes on QEMU's self-exit through S5 (docs/power.md).
|
||||
{"name": "orderly-shutdown",
|
||||
"smp": 4,
|
||||
"timeout": 90,
|
||||
"qmp_after": {"delay": 5, "command": "system_powerdown"},
|
||||
"expect": r"power: button pressed[\s\S]*"
|
||||
r"init: shutting down[\s\S]*"
|
||||
r"power: entering S5",
|
||||
"fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"},
|
||||
# M8: the boot log is persisted to the USB FAT volume. Reuses the orderly-
|
||||
# shutdown build (full tree + power button): init spawns log-flush at boot,
|
||||
# which copies the kernel log to /mnt/usb/DANOS.LOG once /mnt/usb is mounted
|
||||
# (first marker); then the power button drives init's own pre-teardown flush
|
||||
# (second marker), proving both triggers write the file while storage is up.
|
||||
{"name": "log-flush",
|
||||
"build_case": "orderly-shutdown",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_after": {"delay": 8, "command": "system_powerdown"},
|
||||
"expect": r"log-flush: wrote \d+ bytes to /mnt/usb/DANOS\.LOG[\s\S]*"
|
||||
r"init: flushed log to /mnt/usb/DANOS\.LOG[\s\S]*"
|
||||
r"power: entering S5",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
|
||||
# reports its _HID devices — the two PS/2 nodes must appear with resources
|
||||
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/m19-m20-plan.md).
|
||||
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/discovery.md).
|
||||
{"name": "acpi-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -316,9 +434,6 @@ CASES = [
|
||||
{"name": "device-list",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"device-list: \d+ devices[\s\S]*"
|
||||
r"device-list: subscribed[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
@@ -331,9 +446,6 @@ CASES = [
|
||||
{"name": "driver-restart",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci",
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0"],
|
||||
"expect": r"usb-xhci-bus: hello acknowledged[\s\S]*"
|
||||
r"device-manager: restarting crash-test[\s\S]*"
|
||||
r"device-manager: crash-test is failing repeatedly",
|
||||
@@ -341,6 +453,7 @@ CASES = [
|
||||
# The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses
|
||||
# it and spawns each as a ring-3 process (here the VFS-server stub heartbeats).
|
||||
{"name": "initial-ramdisk",
|
||||
"timeout": 60, # the acpi service's boot-time SCI setup can push the marker past 30s under load
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The user-space VFS: a client opens/writes/reads a file through the rt file
|
||||
@@ -354,27 +467,21 @@ CASES = [
|
||||
{"name": "input",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IO passthrough + IRQ-as-IPC: a user-space HPET driver maps device MMIO into
|
||||
# its own address space, binds the device's interrupt to an IPC endpoint, and
|
||||
# is woken by the hardware five times while blocked (never polling).
|
||||
{"name": "hpet",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Device manager: a ring-3 service enumerates /system/devices and matches each
|
||||
# device to a driver (discovery + policy in user space). This increment logs the
|
||||
# decision; spawning follows.
|
||||
# Device manager: a ring-3 service enumerates /system/devices, matches the PCI host
|
||||
# bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn
|
||||
# -> driver-up (the spawned pci-bus logs "<N> functions found").
|
||||
{"name": "device-manager",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Bus driver: a user process claims a device, enumerates its children from the
|
||||
# hardware, and publishes each with dev_register — and the kernel refuses a child
|
||||
# whose window escapes the parent's (else dev_register maps arbitrary memory).
|
||||
{"name": "bus",
|
||||
# device_register containment (in-kernel): registering a child whose MMIO window
|
||||
# escapes its parent's grant is refused (NotContained) — else dev_register would map
|
||||
# arbitrary physical memory — while an identical re-register stays idempotent.
|
||||
{"name": "containment",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||
# keeps its own binding. The path hpet never takes, since it runs forever.
|
||||
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||
{"name": "irqfree",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -385,9 +492,8 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
||||
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
||||
{"name": "poweroff",
|
||||
"expect": r"DANOS-POWER: attempting poweroff",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Soft-off (S5) is owned by the ring-3 acpi service now (see orderly-shutdown);
|
||||
# the kernel keeps only reboot (FADT reset register, no AML).
|
||||
{"name": "reboot",
|
||||
"expect": r"DANOS-POWER: attempting reboot",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -397,31 +503,16 @@ TIMEOUT = 30 # seconds per case
|
||||
|
||||
|
||||
def build(arch, case):
|
||||
cmd = ["zig", "build", f"-Dtest-case={case}"] + arch["zig_flags"]
|
||||
# -Dserial: the harness asserts on markers the kernel writes to serial0, so the
|
||||
# serial log sink must be compiled in. It is off by default (a flashed real-
|
||||
# hardware image keeps its log in RAM instead; see build.zig / serial.zig).
|
||||
cmd = ["zig", "build", f"-Dtest-case={case}", "-Dserial=true"] + arch["zig_flags"]
|
||||
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
|
||||
if r.returncode != 0:
|
||||
return r.stderr.strip() or r.stdout.strip()
|
||||
return None
|
||||
|
||||
|
||||
def make_esp(arch):
|
||||
"""Assemble a fresh EFI System Partition from the freshly built binaries."""
|
||||
esp = os.path.join(WORK, "esp")
|
||||
if os.path.exists(esp):
|
||||
shutil.rmtree(esp)
|
||||
efi_dest, efi_src = arch["efi_app"]
|
||||
kern_dest, kern_src = arch["kernel"]
|
||||
fhs = os.path.join(REPO, "zig-out") # zig-out is the FHS image
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True)
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(kern_dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(fhs, efi_src), os.path.join(esp, efi_dest))
|
||||
shutil.copy(os.path.join(fhs, kern_src), os.path.join(esp, kern_dest))
|
||||
for dest, src in arch.get("extra", []):
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(fhs, src), os.path.join(esp, dest))
|
||||
return esp
|
||||
|
||||
|
||||
def resolve_firmware(arch):
|
||||
"""Collapse the ovmf_code/ovmf_vars candidate lists to the first path that
|
||||
exists on this machine. Mutates `arch` in place; idempotent (a resolved
|
||||
@@ -440,12 +531,37 @@ def resolve_firmware(arch):
|
||||
+ "\nInstall OVMF (edk2-ovmf / ovmf) or add its path above.")
|
||||
|
||||
|
||||
def qmp_send(path, command):
|
||||
"""One QMP command: connect, capabilities handshake, execute. Raises on any
|
||||
failure — the caller retries until the guest's socket is ready. This is how
|
||||
a case injects a host-side event (system_powerdown = the ACPI power button)
|
||||
into the running guest (docs/power.md)."""
|
||||
sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
sock.settimeout(5)
|
||||
try:
|
||||
sock.connect(path)
|
||||
stream = sock.makefile("rw")
|
||||
stream.readline() # the QMP greeting
|
||||
stream.write(json.dumps({"execute": "qmp_capabilities"}) + "\n")
|
||||
stream.flush()
|
||||
stream.readline() # {"return": {}}
|
||||
stream.write(json.dumps({"execute": command}) + "\n")
|
||||
stream.flush()
|
||||
stream.readline()
|
||||
finally:
|
||||
sock.close()
|
||||
|
||||
|
||||
def run_case(arch, case):
|
||||
err = build(arch, case["name"])
|
||||
# A case's kernel build defaults to its name; `build_case` decouples the two
|
||||
# so a case can reuse another's kernel (e.g. usb-boot reuses smoke's).
|
||||
err = build(arch, case.get("build_case", case["name"]))
|
||||
if err:
|
||||
return False, "build failed:\n" + err
|
||||
|
||||
esp = make_esp(arch)
|
||||
# The bootable FAT32 USB image the build produced (tools/make-fat-image.py),
|
||||
# presented to the guest as a usb-storage device (see qemu_args).
|
||||
boot_volume = os.path.join(REPO, "zig-out", "danos-usb.img")
|
||||
vars_fd = os.path.join(WORK, "vars.fd")
|
||||
shutil.copy(arch["ovmf_vars"], vars_fd)
|
||||
serial = os.path.join(WORK, "serial.log")
|
||||
@@ -455,17 +571,35 @@ def run_case(arch, case):
|
||||
expect = re.compile(case["expect"])
|
||||
fail = re.compile(case["fail"]) if case.get("fail") else None
|
||||
|
||||
cmd = [arch["qemu"]] + arch["qemu_args"](arch, esp, vars_fd, serial)
|
||||
cmd = [arch["qemu"]] + arch["qemu_args"](arch, boot_volume, vars_fd, serial)
|
||||
if case.get("smp"): # some cases need more than one core (e.g. parallelism)
|
||||
cmd += ["-smp", str(case["smp"])]
|
||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||
cmd += case["qemu_extra"]
|
||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||
# hook injects host-side events into the guest mid-run. Kept under a short temp
|
||||
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
|
||||
# deep worktree path (e.g. .claude/worktrees/<name>/zig-out/qemu-test/qmp.sock)
|
||||
# blows that limit on macOS, so QEMU fails to bind and exits before booting.
|
||||
qmp_path = os.path.join(tempfile.gettempdir(), f"danos-qmp-{os.getpid()}.sock")
|
||||
if os.path.exists(qmp_path):
|
||||
os.remove(qmp_path)
|
||||
cmd += ["-qmp", f"unix:{qmp_path},server,nowait"]
|
||||
qmp_after = case.get("qmp_after") # {"delay": seconds, "command": "..."}
|
||||
qmp_sent = False
|
||||
started = time.monotonic()
|
||||
qemu = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
try:
|
||||
timeout = case.get("timeout", TIMEOUT)
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(0.2)
|
||||
if qmp_after and not qmp_sent and time.monotonic() - started >= qmp_after["delay"]:
|
||||
try:
|
||||
qmp_send(qmp_path, qmp_after["command"])
|
||||
qmp_sent = True
|
||||
except OSError:
|
||||
pass # socket not up yet; retry next tick
|
||||
text = ""
|
||||
if os.path.exists(serial):
|
||||
with open(serial, "r", errors="replace") as f:
|
||||
@@ -473,6 +607,8 @@ def run_case(arch, case):
|
||||
if fail and fail.search(text):
|
||||
return False, "hit failure marker"
|
||||
if expect.search(text):
|
||||
if qmp_after and not qmp_sent:
|
||||
continue # the hook must deliver before the case may pass
|
||||
return True, "matched " + repr(case["expect"])
|
||||
if qemu.poll() is not None: # QEMU exited on its own
|
||||
if expect.search(text):
|
||||
|
||||
@@ -0,0 +1,362 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Format a real FAT32 image from a set of host files — the danos boot volume.
|
||||
|
||||
Mirrors tools/make-initial-ramdisk.py in spirit: pure Python 3 standard library,
|
||||
no external tools (no mkfs.fat / mtools). It writes a valid FAT32 filesystem — a
|
||||
boot sector + BPB, an FSInfo sector, a backup boot sector, two FATs, and a
|
||||
directory tree of clusters — so UEFI/OVMF boots \\EFI\\BOOT\\BOOTX64.efi off it
|
||||
and the danos FAT driver mounts the same image.
|
||||
|
||||
make-fat-image.py <out.img> <size-MiB> [<dest-path> <host-file>]...
|
||||
make-fat-image.py --verify <out.img>
|
||||
|
||||
Each <dest-path> is a forward-slash path inside the image (e.g.
|
||||
"EFI/BOOT/BOOTX64.efi"); intermediate directories are created. Names that do not
|
||||
fit 8.3 get a mangled short name plus long-file-name (LFN) entries.
|
||||
"""
|
||||
|
||||
import struct
|
||||
import sys
|
||||
|
||||
SECTOR = 512
|
||||
SECTORS_PER_CLUSTER = 1 # 512-byte clusters keep the cluster count high for FAT32
|
||||
RESERVED_SECTORS = 32
|
||||
NUM_FATS = 2
|
||||
CLUSTER_BYTES = SECTOR * SECTORS_PER_CLUSTER
|
||||
|
||||
END_OF_CHAIN = 0x0FFFFFFF
|
||||
BAD_CLUSTER = 0x0FFFFFF7
|
||||
|
||||
ATTR_ARCHIVE = 0x20
|
||||
ATTR_DIRECTORY = 0x10
|
||||
ATTR_LONG_NAME = 0x0F
|
||||
|
||||
VALID_83 = set("ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789$%'-_@~!(){}^#& ")
|
||||
|
||||
|
||||
def fat32_geometry(total_sectors):
|
||||
"""Solve for the FAT size (sectors per FAT) and cluster count that fit."""
|
||||
fat_size = 1
|
||||
while True:
|
||||
data_sectors = total_sectors - RESERVED_SECTORS - NUM_FATS * fat_size
|
||||
cluster_count = data_sectors // SECTORS_PER_CLUSTER
|
||||
needed = ((cluster_count + 2) * 4 + SECTOR - 1) // SECTOR
|
||||
if needed <= fat_size:
|
||||
return fat_size, cluster_count
|
||||
fat_size = needed
|
||||
|
||||
|
||||
class Fat32Image:
|
||||
def __init__(self, total_sectors):
|
||||
self.total_sectors = total_sectors
|
||||
self.fat_size, self.cluster_count = fat32_geometry(total_sectors)
|
||||
if self.cluster_count < 65525:
|
||||
sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters "
|
||||
f"< 65525); use a larger size")
|
||||
self.first_data_sector = RESERVED_SECTORS + NUM_FATS * self.fat_size
|
||||
# The FAT, in memory: entry 0 media, entry 1 EOC, entry 2 the root dir.
|
||||
self.fat = [0] * (self.cluster_count + 2)
|
||||
self.fat[0] = 0x0FFFFFF8
|
||||
self.fat[1] = END_OF_CHAIN
|
||||
self.fat[2] = END_OF_CHAIN
|
||||
self.next_free = 3
|
||||
self.cluster_data = {} # cluster number -> bytes (one cluster's worth)
|
||||
|
||||
def alloc(self):
|
||||
cluster = self.next_free
|
||||
if cluster >= self.cluster_count + 2:
|
||||
sys.exit("error: image out of clusters")
|
||||
self.next_free += 1
|
||||
self.fat[cluster] = END_OF_CHAIN
|
||||
return cluster
|
||||
|
||||
def store_chain(self, content):
|
||||
"""Allocate a cluster chain holding `content` and return its first cluster."""
|
||||
length = max(1, (len(content) + CLUSTER_BYTES - 1) // CLUSTER_BYTES)
|
||||
clusters = [self.alloc() for _ in range(length)]
|
||||
for i in range(length - 1):
|
||||
self.fat[clusters[i]] = clusters[i + 1]
|
||||
for i, cluster in enumerate(clusters):
|
||||
chunk = content[i * CLUSTER_BYTES:(i + 1) * CLUSTER_BYTES]
|
||||
self.cluster_data[cluster] = chunk + b"\x00" * (CLUSTER_BYTES - len(chunk))
|
||||
return clusters[0]
|
||||
|
||||
def store_directory(self, first_cluster, entries):
|
||||
"""Write directory `entries` (bytes) into `first_cluster`, extending the chain."""
|
||||
length = max(1, (len(entries) + CLUSTER_BYTES - 1) // CLUSTER_BYTES)
|
||||
clusters = [first_cluster]
|
||||
for _ in range(length - 1):
|
||||
clusters.append(self.alloc())
|
||||
for i in range(len(clusters) - 1):
|
||||
self.fat[clusters[i]] = clusters[i + 1]
|
||||
for i, cluster in enumerate(clusters):
|
||||
chunk = entries[i * CLUSTER_BYTES:(i + 1) * CLUSTER_BYTES]
|
||||
self.cluster_data[cluster] = chunk + b"\x00" * (CLUSTER_BYTES - len(chunk))
|
||||
|
||||
def cluster_sector(self, cluster):
|
||||
return self.first_data_sector + (cluster - 2) * SECTORS_PER_CLUSTER
|
||||
|
||||
def serialize(self):
|
||||
image = bytearray(self.total_sectors * SECTOR)
|
||||
image[0:SECTOR] = self.boot_sector()
|
||||
image[SECTOR:2 * SECTOR] = self.fsinfo_sector()
|
||||
image[6 * SECTOR:7 * SECTOR] = self.boot_sector() # backup boot sector
|
||||
# Both FATs.
|
||||
fat_bytes = b"".join(struct.pack("<I", entry & 0x0FFFFFFF) for entry in self.fat)
|
||||
fat_bytes += b"\x00" * (self.fat_size * SECTOR - len(fat_bytes))
|
||||
for copy in range(NUM_FATS):
|
||||
base = (RESERVED_SECTORS + copy * self.fat_size) * SECTOR
|
||||
image[base:base + len(fat_bytes)] = fat_bytes
|
||||
# The data region (clusters).
|
||||
for cluster, data in self.cluster_data.items():
|
||||
base = self.cluster_sector(cluster) * SECTOR
|
||||
image[base:base + len(data)] = data
|
||||
return bytes(image)
|
||||
|
||||
def boot_sector(self):
|
||||
sector = bytearray(SECTOR)
|
||||
# BPB.
|
||||
struct.pack_into(
|
||||
"<3s8sHBHBHHBHHHII", sector, 0,
|
||||
b"\xEB\x58\x90", # jump
|
||||
b"MSWIN4.1", # OEM name (widest firmware compatibility)
|
||||
SECTOR, # bytes per sector
|
||||
SECTORS_PER_CLUSTER, # sectors per cluster
|
||||
RESERVED_SECTORS, # reserved sector count
|
||||
NUM_FATS, # number of FATs
|
||||
0, # root entry count (0 for FAT32)
|
||||
0, # total sectors 16 (0 -> use 32)
|
||||
0xF8, # media descriptor
|
||||
0, # FAT size 16 (0 for FAT32)
|
||||
32, # sectors per track
|
||||
2, # heads
|
||||
0, # hidden sectors
|
||||
self.total_sectors, # total sectors 32
|
||||
)
|
||||
# FAT32 extended BPB (offset 36).
|
||||
struct.pack_into(
|
||||
"<IHHIHH12sBBBI11s8s", sector, 36,
|
||||
self.fat_size, # FAT size 32
|
||||
0, # extended flags
|
||||
0, # filesystem version
|
||||
2, # root cluster
|
||||
1, # FSInfo sector
|
||||
6, # backup boot sector
|
||||
b"\x00" * 12, # reserved
|
||||
0x80, # drive number
|
||||
0, # reserved
|
||||
0x29, # extended boot signature
|
||||
0x12345678, # volume id
|
||||
b"DANOS ", # volume label
|
||||
b"FAT32 ", # filesystem type
|
||||
)
|
||||
sector[510] = 0x55
|
||||
sector[511] = 0xAA
|
||||
return bytes(sector)
|
||||
|
||||
def fsinfo_sector(self):
|
||||
sector = bytearray(SECTOR)
|
||||
struct.pack_into("<I", sector, 0, 0x41615252) # lead signature
|
||||
struct.pack_into("<I", sector, 484, 0x61417272) # struct signature
|
||||
free = self.cluster_count - (self.next_free - 2)
|
||||
struct.pack_into("<I", sector, 488, free) # free count
|
||||
struct.pack_into("<I", sector, 492, self.next_free) # next free hint
|
||||
struct.pack_into("<I", sector, 508, 0xAA550000) # trail signature
|
||||
return bytes(sector)
|
||||
|
||||
|
||||
def lfn_checksum(short_name):
|
||||
checksum = 0
|
||||
for byte in short_name:
|
||||
checksum = (((checksum & 1) << 7) + (checksum >> 1) + byte) & 0xFF
|
||||
return checksum
|
||||
|
||||
|
||||
def short_name_for(name, used):
|
||||
"""Return (raw 11-byte 8.3 name, needs_lfn)."""
|
||||
if "." in name and not name.startswith("."):
|
||||
base, ext = name.rsplit(".", 1)
|
||||
else:
|
||||
base, ext = name, ""
|
||||
upper_base, upper_ext = base.upper(), ext.upper()
|
||||
# A name fits 8.3 if it is short enough and uses valid characters; a lowercase
|
||||
# name is simply stored uppercased (FAT is case-insensitive, so the bootloader
|
||||
# and the danos driver still find it). Only genuinely non-8.3 names (too long,
|
||||
# e.g. initial-ramdisk.img) get a mangled short name plus LFN entries.
|
||||
fits = (1 <= len(base) <= 8 and len(ext) <= 3
|
||||
and all(c in VALID_83 for c in upper_base + upper_ext))
|
||||
if fits:
|
||||
return (upper_base.ljust(8) + upper_ext.ljust(3)).encode("ascii"), False
|
||||
# Mangle to STEM~N.EXT.
|
||||
stem = "".join(c for c in upper_base if c in VALID_83 and c != " ")[:6] or "FILE"
|
||||
index = 1
|
||||
while True:
|
||||
candidate = f"{stem}~{index}".ljust(8)[:8] + upper_ext.ljust(3)[:3]
|
||||
raw = candidate.encode("ascii")
|
||||
if raw not in used:
|
||||
used.add(raw)
|
||||
return raw, True
|
||||
index += 1
|
||||
|
||||
|
||||
def lfn_entries(name, short_raw):
|
||||
checksum = lfn_checksum(short_raw)
|
||||
units = list(name.encode("utf-16-le"))
|
||||
pairs = [bytes(units[i:i + 2]) for i in range(0, len(units), 2)]
|
||||
pairs.append(b"\x00\x00") # null terminator
|
||||
while len(pairs) % 13 != 0:
|
||||
pairs.append(b"\xff\xff")
|
||||
count = len(pairs) // 13
|
||||
out = bytearray()
|
||||
for sequence in range(count, 0, -1): # stored last-logical-first
|
||||
piece = pairs[(sequence - 1) * 13:sequence * 13]
|
||||
entry = bytearray(32)
|
||||
entry[0] = sequence | (0x40 if sequence == count else 0)
|
||||
for i in range(5):
|
||||
entry[1 + i * 2:1 + i * 2 + 2] = piece[i]
|
||||
entry[11] = ATTR_LONG_NAME
|
||||
entry[12] = 0
|
||||
entry[13] = checksum
|
||||
for i in range(6):
|
||||
entry[14 + i * 2:14 + i * 2 + 2] = piece[5 + i]
|
||||
entry[26:28] = b"\x00\x00"
|
||||
for i in range(2):
|
||||
entry[28 + i * 2:28 + i * 2 + 2] = piece[11 + i]
|
||||
out += entry
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def short_entry(raw11, attributes, cluster, size):
|
||||
return struct.pack(
|
||||
"<11sBBBHHHHHHHI",
|
||||
raw11, attributes, 0, 0, 0, 0, 0,
|
||||
(cluster >> 16) & 0xFFFF, 0, 0, cluster & 0xFFFF, size,
|
||||
)
|
||||
|
||||
|
||||
def write_directory(image, cluster, children, parent_cluster, is_root):
|
||||
"""Recursively lay out a directory: allocate child clusters, build entries."""
|
||||
entries = bytearray()
|
||||
if not is_root:
|
||||
entries += short_entry(b". ", ATTR_DIRECTORY, cluster, 0)
|
||||
parent = 0 if parent_cluster == 2 else parent_cluster
|
||||
entries += short_entry(b".. ", ATTR_DIRECTORY, parent, 0)
|
||||
used_short_names = set()
|
||||
for name, child in children.items():
|
||||
raw, needs_lfn = short_name_for(name, used_short_names)
|
||||
used_short_names.add(raw)
|
||||
if child["type"] == "dir":
|
||||
child_cluster = image.alloc()
|
||||
if needs_lfn:
|
||||
entries += lfn_entries(name, raw)
|
||||
entries += short_entry(raw, ATTR_DIRECTORY, child_cluster, 0)
|
||||
write_directory(image, child_cluster, child["children"], cluster, False)
|
||||
else:
|
||||
data = child["data"]
|
||||
first = image.store_chain(data) if data else 0
|
||||
if needs_lfn:
|
||||
entries += lfn_entries(name, raw)
|
||||
entries += short_entry(raw, ATTR_ARCHIVE, first, len(data))
|
||||
image.store_directory(cluster, bytes(entries))
|
||||
|
||||
|
||||
def build_tree(pairs):
|
||||
root = {}
|
||||
for dest, host in pairs:
|
||||
with open(host, "rb") as handle:
|
||||
data = handle.read()
|
||||
parts = [p for p in dest.replace("\\", "/").split("/") if p]
|
||||
node = root
|
||||
for part in parts[:-1]:
|
||||
node = node.setdefault(part, {"type": "dir", "children": {}})["children"]
|
||||
node[parts[-1]] = {"type": "file", "data": data}
|
||||
return root
|
||||
|
||||
|
||||
def build(out_path, size_mib, pairs):
|
||||
total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
image = Fat32Image(total_sectors)
|
||||
tree = build_tree(pairs)
|
||||
write_directory(image, 2, tree, 0, True)
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(image.serialize())
|
||||
print(f"make-fat-image: wrote {out_path} "
|
||||
f"({size_mib} MiB FAT32, {image.cluster_count} clusters)")
|
||||
|
||||
|
||||
def verify(path):
|
||||
with open(path, "rb") as handle:
|
||||
data = handle.read()
|
||||
if len(data) < SECTOR or data[510] != 0x55 or data[511] != 0xAA:
|
||||
sys.exit("verify: missing 0x55AA boot signature")
|
||||
bytes_per_sector, sectors_per_cluster = struct.unpack_from("<HB", data, 11)
|
||||
reserved, num_fats = struct.unpack_from("<H", data, 14)[0], data[16]
|
||||
fat_size_32, root_cluster = struct.unpack_from("<I", data, 36)[0], struct.unpack_from("<I", data, 44)[0]
|
||||
total_sectors = struct.unpack_from("<I", data, 32)[0]
|
||||
if bytes_per_sector != SECTOR or sectors_per_cluster == 0 or num_fats == 0 or fat_size_32 == 0:
|
||||
sys.exit("verify: implausible BPB")
|
||||
first_data = reserved + num_fats * fat_size_32
|
||||
cluster_count = (total_sectors - first_data) // sectors_per_cluster
|
||||
if cluster_count < 65525:
|
||||
sys.exit(f"verify: not FAT32 ({cluster_count} clusters)")
|
||||
# Resolve EFI/BOOT/BOOTX64.efi through the directory tree to prove it is present.
|
||||
if not _resolve(data, ["EFI", "BOOT", "BOOTX64.EFI"], root_cluster,
|
||||
reserved, num_fats, fat_size_32, first_data, sectors_per_cluster):
|
||||
sys.exit("verify: EFI/BOOT/BOOTX64.efi not found")
|
||||
print(f"verify: {path} is FAT32 ({cluster_count} clusters); EFI/BOOT/BOOTX64.efi present")
|
||||
|
||||
|
||||
def _read_fat(data, cluster, reserved):
|
||||
offset = reserved * SECTOR + cluster * 4
|
||||
return struct.unpack_from("<I", data, offset)[0] & 0x0FFFFFFF
|
||||
|
||||
|
||||
def _resolve(data, parts, cluster, reserved, num_fats, fat_size, first_data, spc):
|
||||
for part in parts:
|
||||
cluster = _find(data, cluster, part, reserved, first_data, spc)
|
||||
if cluster is None:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _find(data, dir_cluster, name, reserved, first_data, spc):
|
||||
target = name.upper()
|
||||
cluster = dir_cluster
|
||||
guard = 0
|
||||
while cluster >= 2 and cluster < BAD_CLUSTER and guard < 100000:
|
||||
sector = first_data + (cluster - 2) * spc
|
||||
for s in range(spc):
|
||||
base = (sector + s) * SECTOR
|
||||
for i in range(SECTOR // 32):
|
||||
entry = data[base + i * 32:base + i * 32 + 32]
|
||||
if entry[0] == 0x00:
|
||||
return None
|
||||
if entry[0] == 0xE5 or (entry[11] & ATTR_LONG_NAME) == ATTR_LONG_NAME:
|
||||
continue
|
||||
raw = entry[0:11]
|
||||
short = (raw[0:8].rstrip().decode("latin1") +
|
||||
("." + raw[8:11].rstrip().decode("latin1") if raw[8:11].strip() else "")).upper()
|
||||
if short == target:
|
||||
return ((entry[20] | (entry[21] << 8)) << 16) | (entry[26] | (entry[27] << 8))
|
||||
cluster = _read_fat(data, cluster, reserved)
|
||||
guard += 1
|
||||
return None
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
if len(argv) < 3 or (len(argv) - 3) % 2 != 0:
|
||||
sys.exit("usage: make-fat-image.py <out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
" make-fat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
rest = argv[3:]
|
||||
pairs = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||
build(out_path, size_mib, pairs)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
Executable
+22
@@ -0,0 +1,22 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# sort-lines-group-by-start — cluster lines that share their first
|
||||
# whitespace-separated field ($1). Keys appear in first-seen order, and lines
|
||||
# within a key keep their original order. It groups; it does NOT sort.
|
||||
#
|
||||
# Pass the log file as an argument; result is written to stdout:
|
||||
#
|
||||
# tools/sort-lines-group-by-start.sh filename.log
|
||||
#
|
||||
# Useful for a serial/boot log where several sources interleave and each line is
|
||||
# prefixed with its source (the first field): this pulls every source's lines
|
||||
# back together, in the order the sources first appeared, without reordering
|
||||
# within a source.
|
||||
#
|
||||
# input output
|
||||
# pci-bus: scan start pci-bus: scan start
|
||||
# acpi: reported PNP0303 pci-bus: 5 functions
|
||||
# pci-bus: 5 functions acpi: reported PNP0303
|
||||
# acpi: reported PNP0501 acpi: reported PNP0501
|
||||
|
||||
awk '{lines[$1] = lines[$1] ? lines[$1] ORS $0 : $0; if (!seen[$1]++) order[++count] = $1} END {for (i=1; i<=count; i++) print lines[order[i]]}' "$@"
|
||||
Reference in New Issue
Block a user