Compare commits
208
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5ab7263c9c | ||
|
|
7f415e724f | ||
|
|
cf140eb772 | ||
|
|
e2dddc941f | ||
|
|
28b3635979 | ||
|
|
6101e429ba | ||
|
|
a4e44e8f31 | ||
|
|
c7e9b5a4f6 | ||
|
|
6bc329456a | ||
|
|
ed3b3f1c45 | ||
|
|
8259678f0a | ||
|
|
f4813c8e99 | ||
|
|
8b7f1d009c | ||
|
|
def34e71fc | ||
|
|
1b33f48acd | ||
|
|
b3a8147bd7 | ||
|
|
0730e77530 | ||
|
|
73df864fd2 | ||
|
|
11e363896f | ||
|
|
6e8b02d771 | ||
|
|
d26515706e | ||
|
|
acf8ff2c33 | ||
|
|
e5dcc9790b | ||
|
|
f8ad4ac971 | ||
|
|
914af52b94 | ||
|
|
bdd8a48476 | ||
|
|
f309ce04f4 | ||
|
|
9989ebbec7 | ||
|
|
626e3c5e9b | ||
|
|
2c63c76288 | ||
|
|
05fc1764de | ||
|
|
65bb04d890 | ||
|
|
24c49f56e1 | ||
|
|
0217662808 | ||
|
|
4231301896 | ||
|
|
58927ed7e5 | ||
|
|
6e0e0a62c6 | ||
|
|
88ad432758 | ||
|
|
9333d0572f | ||
|
|
f3342118f5 | ||
|
|
2723b6f778 | ||
|
|
a01a4f3b3d | ||
|
|
e1605e3235 | ||
|
|
1e80c57484 | ||
|
|
c3e9c59086 | ||
|
|
88ed3c5417 | ||
|
|
3fc2d5b083 | ||
|
|
105203b447 | ||
|
|
f9cf0007c5 | ||
|
|
69b018cc32 | ||
|
|
cd812cc00e | ||
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f | ||
|
|
67702fa250 | ||
|
|
8a38540312 | ||
|
|
54635eecf5 | ||
|
|
184d90c2c6 | ||
|
|
a32eed877d | ||
|
|
347a041d85 | ||
|
|
53e42837e0 | ||
|
|
f52c591f5e | ||
|
|
77d2e22ed1 | ||
|
|
a64a01a6a9 | ||
|
|
35e8921de8 | ||
|
|
3fb9d5936a | ||
|
|
452080e997 | ||
|
|
5b63a841ba | ||
|
|
4b9507bd59 | ||
|
|
7dec1b0767 | ||
|
|
45b8fd8614 | ||
|
|
dd22bfbc48 | ||
|
|
6e60daed6a | ||
|
|
446f655c69 | ||
|
|
a785efa4a3 | ||
|
|
767a2a9a7c | ||
|
|
dd044fb115 | ||
|
|
3ec14509a0 | ||
|
|
1f2c60b3ec | ||
|
|
d71a5f25d3 | ||
|
|
2a0f17ae86 | ||
|
|
9ef61a0844 | ||
|
|
688b9101e8 | ||
|
|
1d7ba814dc | ||
|
|
8aba86b4ce | ||
|
|
a0c83f4b3f | ||
|
|
1ea48ed5d6 | ||
|
|
dfc7d6a609 | ||
|
|
77a3ccd33d | ||
|
|
07da27dc39 | ||
|
|
d89657d0a4 | ||
|
|
849b4b62d4 | ||
|
|
8589bf713b | ||
|
|
738f6aa697 | ||
|
|
01e56e3f36 | ||
|
|
d5d15cefcb | ||
|
|
fd96a35eb9 | ||
|
|
e3fe3f3f45 | ||
|
|
60da667b42 | ||
|
|
36145e623b | ||
|
|
565415327d | ||
|
|
bf6bdb389d | ||
|
|
0628944b15 | ||
|
|
e6d0bb7ef0 | ||
|
|
5ca804d827 | ||
|
|
a299363b59 | ||
|
|
d8dd62c639 | ||
|
|
d106b6e8dc | ||
|
|
af2c766f42 | ||
|
|
d26262bf56 | ||
|
|
10b89c06ff | ||
|
|
a2a05d0b3d | ||
|
|
75d62660b0 | ||
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 | ||
|
|
650a1b1595 | ||
|
|
d8c55c6f2f | ||
|
|
2ebfb0c3b0 | ||
|
|
888eaa74e1 | ||
|
|
ed76cbbc79 | ||
|
|
140229b88d | ||
|
|
cb2379fd06 | ||
|
|
1665b239b0 | ||
|
|
70ed0337f8 | ||
|
|
116b8f6c41 | ||
|
|
77901bbba6 | ||
|
|
78582d24d2 | ||
|
|
1cdffe21b1 | ||
|
|
4df90bc212 | ||
|
|
713e77354b | ||
|
|
abb7b1b634 | ||
|
|
f5f0e15769 | ||
|
|
8652b4a724 | ||
|
|
e8233127c7 | ||
|
|
88e92254e9 | ||
|
|
5725d35e5b | ||
|
|
8c95525793 | ||
|
|
5bba5d3363 | ||
|
|
80b72db676 | ||
|
|
aa0c97353a | ||
|
|
dd93204b44 | ||
|
|
c7b17aaa0e | ||
|
|
d7a154a596 | ||
|
|
1bf91115dd | ||
|
|
65244e3103 | ||
|
|
75ccfff171 | ||
|
|
2a583d55a8 | ||
|
|
d218d93f79 | ||
|
|
a5fe63c1dd | ||
|
|
6b3ae0c997 | ||
|
|
59104dd988 | ||
|
|
e499f500c3 | ||
|
|
2ab0d129a2 | ||
|
|
d702d2e9ae | ||
|
|
3ec2d1828a | ||
|
|
21657943a8 | ||
|
|
e612d948d2 | ||
|
|
4ef21fa083 | ||
|
|
125a3b4993 | ||
|
|
e7c7e7b94c | ||
|
|
a581712b09 | ||
|
|
8eb4210251 | ||
|
|
56110b0019 | ||
|
|
afbf10f7fc | ||
|
|
b61b7775b9 | ||
|
|
be81394be3 | ||
|
|
47610e8ee2 | ||
|
|
193fd71a50 | ||
|
|
1c2b3ae64d | ||
|
|
d19a0ae38d | ||
|
|
3d1de37d0e | ||
|
|
ceacc6b514 | ||
|
|
8754d4e46a | ||
|
|
15b70856c9 | ||
|
|
83881641ca | ||
|
|
b0f894f50c | ||
|
|
750a73f050 | ||
|
|
eabb81684a | ||
|
|
0a8c81b17a | ||
|
|
9316f9f1c3 | ||
|
|
7384a730be | ||
|
|
b9d9e1e523 | ||
|
|
37fb3cb0cf | ||
|
|
5d57e7b01c | ||
|
|
8a7235c725 | ||
|
|
f4590fcc19 | ||
|
|
bb7597ea0b | ||
|
|
f57a73e8a1 | ||
|
|
724de7bbd0 | ||
|
|
75bc429aa1 | ||
|
|
91aff2bc4a | ||
|
|
546dd44a2a | ||
|
|
7501bd1703 | ||
|
|
1b47de5058 |
@@ -0,0 +1,16 @@
|
||||
# EditorConfig: https://editorconfig.org/
|
||||
# Follows the Zig style guide: https://ziglang.org/documentation/0.16.0/#Style-Guide
|
||||
|
||||
root = true
|
||||
|
||||
[*]
|
||||
charset = utf-8
|
||||
end_of_line = lf
|
||||
indent_style = space
|
||||
indent_size = 4
|
||||
trim_trailing_whitespace = true
|
||||
insert_final_newline = true
|
||||
|
||||
[*.zig]
|
||||
# "Line length: aim for 100; use common sense."
|
||||
max_line_length = 100
|
||||
@@ -0,0 +1 @@
|
||||
*.zig text eol=lf
|
||||
@@ -4,3 +4,6 @@ zig-out/
|
||||
|
||||
# JetBrains IDE
|
||||
.idea/
|
||||
|
||||
.claude/
|
||||
.github/
|
||||
@@ -1 +0,0 @@
|
||||
0.16.0
|
||||
@@ -1,13 +1,32 @@
|
||||
# DanOS
|
||||
Codename: Shodan
|
||||
Version: 1
|
||||
|
||||
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`)
|
||||
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`).
|
||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
||||
channels. See [`docs/`](docs/README.md) for how each piece works.
|
||||
**Codename: Shodan**
|
||||
|
||||
A very small resilient operating system.
|
||||
|
||||
## Zen of DanOS:
|
||||
|
||||
- Resilient Micro-Kernel Architecture.
|
||||
- Every process run in an isolated user space not kernel space.
|
||||
- Processes cannot take down the entire OS with it when they die or is killed
|
||||
- Stable public runtime library, private OS ABI.
|
||||
- Keeps a stable runtime for user space processes between OS versions (great for backwards compatibility)
|
||||
- Allows the underlying OS to be changed without effecting applications
|
||||
- Provides a boundary to enable compatibility between OS's e.g. POSIX, MUSL etc
|
||||
- Drivers are just isolated processes in user space.
|
||||
- Thin binaries that can be restarted like applications.
|
||||
- Useful during driver development.
|
||||
- Drivers can claim MMIO / ports
|
||||
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||
- Drivers can also hook into the process lifecycle to clean up or reset hardware
|
||||
- No legacy to deal with
|
||||
- Zig code uses a clean coding style (Zen of Zig)
|
||||
- Favor reading code over writing code.
|
||||
- No magic numbers.
|
||||
- No shortened names unless its for ABI compatibility or acronyms
|
||||
- Inter-Process Communication (IPC)
|
||||
- Publish and subscribe to Asynchronous Messages
|
||||
- Talk to services and processes synchronously
|
||||
|
||||
## Prerequisites
|
||||
|
||||
@@ -30,8 +49,21 @@ channels. See [`docs/`](docs/README.md) for how each piece works.
|
||||
zig build
|
||||
```
|
||||
|
||||
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
|
||||
(`zig-out/bin/kernel`).
|
||||
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Release media
|
||||
|
||||
```sh
|
||||
zig build release-x86-64
|
||||
```
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
@@ -58,9 +90,13 @@ straight into CI.
|
||||
|
||||
## Documentation
|
||||
|
||||
Design notes explaining the *why* behind the code live in
|
||||
Design notes explaining *why* behind the code live in
|
||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||
|
||||
For the hardware needed to run DanOS — minimum specs plus a plain-language guide
|
||||
matching Intel/AMD CPU generations by name — see
|
||||
[`docs/system-requirements.md`](docs/system-requirements.md).
|
||||
|
||||
## Logo
|
||||
|
||||
San Serif Text "Dan OS" with a black karate belt around it.
|
||||
|
||||
+277
-55
@@ -1,15 +1,25 @@
|
||||
const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const BootInfo = danos.BootInfo;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
const EdidActive = uefi.protocol.edid.Active;
|
||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
|
||||
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
|
||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
||||
// The boot volume is the FHS-shaped zig-out (see build.zig / docs/README.md), so the
|
||||
// loader reads each artifact from its addressed FHS path. UEFI paths use backslashes;
|
||||
// the FAT driver walks the components itself, so no per-directory dance is needed.
|
||||
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The init program: /system/services/init.
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||
|
||||
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -20,7 +30,7 @@ pub fn main() uefi.Status {
|
||||
// report the reason (boot services are still up) and park the machine so the
|
||||
// message stays on screen.
|
||||
boot() catch |err| {
|
||||
log("\r\ndanos: boot failed: ");
|
||||
log("\r\nEFI: boot failed: ");
|
||||
logBytes(@errorName(err));
|
||||
log("\r\n");
|
||||
while (true) asm volatile ("hlt");
|
||||
@@ -33,10 +43,10 @@ fn boot() !noreturn {
|
||||
|
||||
// Everything the kernel needs must be gathered *before* we exit boot
|
||||
// services, since afterwards none of these calls are usable.
|
||||
var boot_info: BootInfo = .{
|
||||
var boot_information: BootInformation = .{
|
||||
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||
.framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{
|
||||
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
|
||||
.base = 0,
|
||||
.width = 0,
|
||||
.height = 0,
|
||||
@@ -52,16 +62,38 @@ fn boot() !noreturn {
|
||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||
};
|
||||
|
||||
const entry = try loadKernel(bs, &boot_info);
|
||||
const entry = try loadKernel(bs, &boot_information);
|
||||
|
||||
log("danos: kernel loaded, exiting boot services\r\n");
|
||||
boot_info.memory_map = try exitBootServices(bs);
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system/services/init (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Hand control to the kernel. `danos.kernel_abi` is SysV, so the pointer is
|
||||
// passed in RDI as the kernel expects — not RCX, which this UEFI binary's
|
||||
// default `.c` convention (Microsoft x64) would use.
|
||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn = @ptrFromInt(entry);
|
||||
kernel(&boot_info);
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("EFI: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
|
||||
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||
// now, while boot services (and the memory map) are still stable — nothing
|
||||
// is allocatable after ExitBootServices, and any allocation between fetching
|
||||
// the map and exiting would invalidate the map key.
|
||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||
|
||||
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||
boot_information.memory_map = try exitBootServices(bs);
|
||||
|
||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||
// We load RDI explicitly (SystemV first arg) rather than trusting this UEFI
|
||||
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
||||
// higher-half) entry — the bootstrap tables map both the low loader code
|
||||
// executing this and the kernel's link address.
|
||||
handoff(cr3, entry, &boot_information);
|
||||
}
|
||||
|
||||
/// A display resolution in pixels.
|
||||
@@ -69,7 +101,7 @@ const Resolution = struct { width: u32, height: u32 };
|
||||
|
||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||
/// and read the resulting graphics mode into our own framebuffer description.
|
||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
||||
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
||||
// which is what tells us the panel's native resolution.
|
||||
@@ -101,7 +133,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
||||
|
||||
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
||||
/// layout we can paint into, so they're rejected.
|
||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !danos.PixelFormat {
|
||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
|
||||
return switch (fmt) {
|
||||
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
||||
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
||||
@@ -166,7 +198,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
||||
|
||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||
/// load its segments, and return the physical entry-point address.
|
||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
@@ -195,13 +227,190 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
||||
read_total += n;
|
||||
}
|
||||
|
||||
return loadElf(bs, image, boot_info);
|
||||
return loadElf(bs, image, boot_information);
|
||||
}
|
||||
|
||||
// --- bootstrap page tables -------------------------------------------------
|
||||
// The kernel is (or will be) linked in the higher half but loaded low; the
|
||||
// firmware's identity map doesn't cover the higher half, so the loader builds
|
||||
// the first set of real page tables and switches CR3 before jumping in. They
|
||||
// carry: an identity map of low RAM (so the loader's own code/stack executing
|
||||
// the switch stays valid, and the low-linked kernel keeps working during the
|
||||
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
|
||||
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
|
||||
// The kernel later builds its own precise tables (paging.init) and abandons
|
||||
// these; they leak as reserved LoaderData (~a handful of frames).
|
||||
|
||||
const pte_present: u64 = 1 << 0;
|
||||
const pte_write: u64 = 1 << 1;
|
||||
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
||||
const pte_address: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
const gib: u64 = 1 << 30;
|
||||
|
||||
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
||||
const TablePool = struct {
|
||||
base: usize,
|
||||
next: usize,
|
||||
cap: usize,
|
||||
|
||||
fn alloc(self: *TablePool) !u64 {
|
||||
if (self.next >= self.cap) return error.OutOfBootstrapFrames;
|
||||
const frame = self.base + self.next * page_size;
|
||||
self.next += 1;
|
||||
@memset(@as(*[512]u64, @ptrFromInt(frame)), 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
fn table(physical: u64) *[512]u64 {
|
||||
return @ptrFromInt(physical);
|
||||
}
|
||||
|
||||
/// Return the next-level table an entry points at, creating it if absent.
|
||||
fn descend(self: *TablePool, entry: *u64) !u64 {
|
||||
if (entry.* & pte_present != 0) return entry.* & pte_address;
|
||||
const frame = try self.alloc();
|
||||
entry.* = frame | pte_present | pte_write;
|
||||
return frame;
|
||||
}
|
||||
|
||||
fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||
const pdpt = try self.descend(pml4e);
|
||||
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = try self.descend(pdpte);
|
||||
table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
||||
}
|
||||
|
||||
fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||
const pdpt = try self.descend(pml4e);
|
||||
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = try self.descend(pdpte);
|
||||
const pde = &table(pd)[(virtual >> 21) & 0x1FF];
|
||||
const pt = try self.descend(pde);
|
||||
table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write;
|
||||
}
|
||||
};
|
||||
|
||||
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
||||
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
||||
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 {
|
||||
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
||||
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
||||
const pool_pages = 64;
|
||||
const block = try bs.allocatePages(.any, .loader_data, pool_pages);
|
||||
var pool = TablePool{ .base = @intFromPtr(block.ptr), .next = 0, .cap = pool_pages };
|
||||
|
||||
const pml4 = try pool.alloc();
|
||||
|
||||
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
||||
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
||||
// framebuffer above 4 GiB would extend this — see the fb window below.
|
||||
var address: u64 = 0;
|
||||
while (address < 4 * gib) : (address += 2 << 20) {
|
||||
try pool.map2M(pml4, address, address); // identity
|
||||
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
|
||||
}
|
||||
|
||||
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||
// pages (the kernel touches fb.base before it builds its own tables).
|
||||
const fb = boot_information.framebuffer;
|
||||
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
||||
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
||||
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||
while (p < fb_end) : (p += 2 << 20) {
|
||||
try pool.map2M(pml4, p, p);
|
||||
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
|
||||
}
|
||||
}
|
||||
|
||||
// Higher-half kernel segments (virtual != physical). While the kernel still links
|
||||
// low its segments sit in the identity range and need no separate mapping
|
||||
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||
// only map segments that actually live in the higher half.
|
||||
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
|
||||
var off: u64 = 0;
|
||||
while (off < seg.pages * page_size) : (off += page_size) {
|
||||
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||
}
|
||||
}
|
||||
|
||||
return pml4;
|
||||
}
|
||||
|
||||
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI,
|
||||
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
||||
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
||||
/// load; the jump target is mapped (identity while low, higher-half once high).
|
||||
fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn {
|
||||
asm volatile (
|
||||
\\cli
|
||||
\\movq %[cr3], %%cr3
|
||||
\\movq %[bi], %%rdi
|
||||
\\callq *%[entry]
|
||||
:
|
||||
: [cr3] "r" (cr3),
|
||||
[bi] "r" (boot_information),
|
||||
[entry] "r" (entry),
|
||||
: .{ .memory = true });
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||
/// and reads from there. Returns the buffer (pointer + length).
|
||||
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
|
||||
return error.NoFileSystem;
|
||||
|
||||
const root = try fs.openVolume();
|
||||
defer _ = root.close() catch {};
|
||||
|
||||
const file = try root.open(name, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
|
||||
try file.setPosition(seek_end);
|
||||
const size: usize = @intCast(try file.getPosition());
|
||||
try file.setPosition(0);
|
||||
if (size == 0) return error.EmptyFile;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||
|
||||
var read_total: usize = 0;
|
||||
while (read_total < size) {
|
||||
const n = try file.read(image[read_total..]);
|
||||
if (n == 0) return error.UnexpectedEof;
|
||||
read_total += n;
|
||||
}
|
||||
return image[0..size];
|
||||
}
|
||||
|
||||
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
progress("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
progress("EFI: initial_ramdisk loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
/// record each segment's layout so the kernel can re-map itself with the right
|
||||
/// permissions.
|
||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize {
|
||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||
|
||||
@@ -218,7 +427,9 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
||||
|
||||
// Reserve the exact physical pages this segment is linked at. This
|
||||
// requires the segment's p_paddr to be free in the firmware memory map;
|
||||
// if it collides, adjust `image_base` in build.zig.
|
||||
// if it collides, adjust `image_base` in build.zig. (Once the kernel
|
||||
// links high — M2 step 4 — p_paddr becomes a separate low load address
|
||||
// via the linker's AT(), and this stays a valid physical allocation.)
|
||||
const mem_sz: usize = @intCast(phdr.p_memsz);
|
||||
const pages = (mem_sz + page_size - 1) / page_size;
|
||||
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
||||
@@ -231,15 +442,18 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
||||
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
||||
@memset(bytes[file_sz..mem_sz], 0);
|
||||
|
||||
// Record it (identity-loaded: virtual == physical) for the kernel's VMM.
|
||||
const n = boot_info.kernel_segment_count;
|
||||
if (n < boot_info.kernel_segments.len) {
|
||||
boot_info.kernel_segments[n] = .{
|
||||
.virt = phdr.p_vaddr,
|
||||
// Record the virtual link address and the physical load address so the
|
||||
// kernel can map itself with the right permissions post-switch. They're
|
||||
// equal while the kernel links low; they diverge once it links high.
|
||||
const n = boot_information.kernel_segment_count;
|
||||
if (n < boot_information.kernel_segments.len) {
|
||||
boot_information.kernel_segments[n] = .{
|
||||
.virtual = phdr.p_vaddr,
|
||||
.physical = phdr.p_paddr,
|
||||
.pages = pages,
|
||||
.flags = phdr.p_flags,
|
||||
};
|
||||
boot_info.kernel_segment_count = n + 1;
|
||||
boot_information.kernel_segment_count = n + 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -250,27 +464,27 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
||||
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
||||
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
||||
/// the exit, so the returned map stays valid for the kernel.
|
||||
fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
||||
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 8) : (attempts += 1) {
|
||||
const info = try bs.getMemoryMapInfo();
|
||||
// Spare descriptors to absorb the growth from the allocations below.
|
||||
const cap = info.len + 8;
|
||||
const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||
const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
||||
const map = bs.getMemoryMap(map_buf) catch {
|
||||
_ = bs.freePool(map_buf.ptr) catch {};
|
||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
||||
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
|
||||
const map = bs.getMemoryMap(map_buffer) catch {
|
||||
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||
continue;
|
||||
};
|
||||
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
||||
_ = bs.freePool(map_buf.ptr) catch {};
|
||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
||||
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||
continue;
|
||||
};
|
||||
// Boot services are gone; do not touch `bs` again. Converting the map is
|
||||
// pure computation on memory we already hold, so it's safe here.
|
||||
return convertMemoryMap(map, regions_buf);
|
||||
return convertMemoryMap(map, regions_buffer);
|
||||
}
|
||||
return error.ExitBootServicesFailed;
|
||||
}
|
||||
@@ -279,8 +493,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
||||
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
||||
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
||||
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
const regions: [*]danos.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
|
||||
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||
// We're about to call boot-services memory `usable`, but our own stack lives
|
||||
// in it and the kernel starts out running on it. Keep the region holding the
|
||||
// current stack pointer reserved so it's never handed out.
|
||||
@@ -297,16 +511,16 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
if (d.number_of_pages == 0) continue;
|
||||
var kind = classify(d);
|
||||
// The descriptor we're executing on stays reserved (see rsp above).
|
||||
const region_end = d.physical_start + d.number_of_pages * danos.page_size;
|
||||
const region_end = d.physical_start + d.number_of_pages * page_size;
|
||||
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
||||
|
||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||
if (count > 0) {
|
||||
const prev = ®ions[count - 1];
|
||||
if (prev.kind == kind and
|
||||
prev.base + prev.pages * danos.page_size == d.physical_start)
|
||||
const previous = ®ions[count - 1];
|
||||
if (previous.kind == kind and
|
||||
previous.base + previous.pages * page_size == d.physical_start)
|
||||
{
|
||||
prev.pages += d.number_of_pages;
|
||||
previous.pages += d.number_of_pages;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -322,7 +536,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
|
||||
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
||||
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
||||
/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio`
|
||||
/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio`
|
||||
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
||||
/// and such holes, and the cache attribute is what actually tells them apart.
|
||||
///
|
||||
@@ -331,7 +545,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
||||
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
||||
/// kernel image and these buffers) lands there and stays reserved.
|
||||
fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
||||
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
|
||||
if (!d.attribute.wb) return .mmio;
|
||||
return switch (d.type) {
|
||||
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
||||
@@ -343,33 +557,41 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
||||
}
|
||||
|
||||
/// Write a compile-time string to the console (best effort).
|
||||
fn log(comptime msg: []const u8) void {
|
||||
fn log(comptime message: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {};
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||
}
|
||||
|
||||
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||
/// `log` directly and always show, so a failed boot still explains itself.
|
||||
fn progress(comptime message: []const u8) void {
|
||||
if (!build_options.serial) return;
|
||||
log(message);
|
||||
}
|
||||
|
||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||
fn logBytes(bytes: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
var buf: [128]u16 = undefined;
|
||||
var buffer: [128]u16 = undefined;
|
||||
var i: usize = 0;
|
||||
for (bytes) |b| {
|
||||
if (i + 1 >= buf.len) break;
|
||||
buf[i] = b;
|
||||
if (i + 1 >= buffer.len) break;
|
||||
buffer[i] = b;
|
||||
i += 1;
|
||||
}
|
||||
buf[i] = 0;
|
||||
_ = out.outputString(buf[0..i :0].ptr) catch {};
|
||||
buffer[i] = 0;
|
||||
_ = out.outputString(buffer[0..i :0].ptr) catch {};
|
||||
}
|
||||
|
||||
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
||||
const table_entries = uefi.system_table.number_of_table_entries;
|
||||
const config_tables = uefi.system_table.configuration_table;
|
||||
const configuration_tables = uefi.system_table.configuration_table;
|
||||
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
||||
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
||||
|
||||
for (0..table_entries) |i| {
|
||||
const entry = config_tables[i];
|
||||
const entry = configuration_tables[i];
|
||||
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
||||
return entry.vendor_table;
|
||||
}
|
||||
@@ -48,53 +48,416 @@ fn timestamp(b: *std.Build) []const u8 {
|
||||
});
|
||||
}
|
||||
|
||||
/// Build one user-space binary the same way for every program (init, and later
|
||||
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
||||
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
||||
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// library/runtime/root.zig, which supplies the root declarations (`main`
|
||||
/// re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary imports.
|
||||
fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, false);
|
||||
}
|
||||
|
||||
/// As `addUserBinary`, but built multi-threaded (`single_threaded = false`) so real
|
||||
/// atomics/TLS work — required before a binary may call `runtime.Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
fn addThreadedUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, true);
|
||||
}
|
||||
|
||||
fn addUserBinaryImpl(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
threaded: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Settings (target, optimize, code model, ...) live on the root module only;
|
||||
// the program and runtime modules leave theirs null and inherit them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
},
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/root.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = !threaded, // a threaded binary needs real atomics/TLS
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The `program` module of a binary built by `addUserBinary` — the module rooted
|
||||
/// at the program's own source file. Per-binary imports (protocol modules, bus
|
||||
/// ABIs) go here, not on the root shim: module imports are not transitive, so an
|
||||
/// import added to the root would be invisible to the program's code.
|
||||
fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||
return exe.root_module.import_table.get("program").?;
|
||||
}
|
||||
|
||||
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||
const KernelModules = struct {
|
||||
boot_handoff: *std.Build.Module,
|
||||
abi: *std.Build.Module,
|
||||
device_abi: *std.Build.Module,
|
||||
architecture: *std.Build.Module,
|
||||
platform: *std.Build.Module,
|
||||
parameters: *std.Build.Module,
|
||||
initial_ramdisk: *std.Build.Module,
|
||||
};
|
||||
|
||||
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||
/// build option baked into `build_options`.
|
||||
fn addKernel(
|
||||
b: *std.Build,
|
||||
kernel_target: std.Build.ResolvedTarget,
|
||||
optimize: std.builtin.OptimizeMode,
|
||||
modules: KernelModules,
|
||||
test_case: ?[]const u8,
|
||||
serial: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
build_options.addOption(bool, "serial", serial);
|
||||
const build_options_module = build_options.createModule();
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||
.{ .name = "abi", .module = modules.abi },
|
||||
.{ .name = "device-abi", .module = modules.device_abi },
|
||||
.{ .name = "architecture", .module = modules.architecture },
|
||||
.{ .name = "platform", .module = modules.platform },
|
||||
.{ .name = "parameters", .module = modules.parameters },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||
/// bundle its own serial kernel while sharing the loader, init, and ramdisk — all
|
||||
/// built once per invocation (the loader's boot breadcrumbs and init's heartbeat
|
||||
/// both follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
init_bin: std.Build.LazyPath,
|
||||
initial_ramdisk_img: std.Build.LazyPath,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_bin);
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
return fat_image;
|
||||
}
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
ensureZigVersion();
|
||||
|
||||
const target = b.standardTargetOptions(.{});
|
||||
const optimize = b.standardOptimizeOption(.{});
|
||||
|
||||
// Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set,
|
||||
// so the module inherits the target of whichever binary imports it — the
|
||||
// freestanding kernel or the UEFI bootloader.
|
||||
const mod = b.addModule("danos", .{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
// The three shared contracts, each with its own audience so every import
|
||||
// declares which one it speaks (no target is set, so each inherits the target of
|
||||
// whichever binary imports it). See docs/coding-standards.md.
|
||||
// boot-handoff : loader <-> kernel (BootInformation, framebuffer, VM layout)
|
||||
// abi : kernel <-> runtime, core (SystemCall, mmap prot flags, page_size)
|
||||
// device-abi : kernel <-> user, devices (DeviceDescriptor, DeviceClass, ...)
|
||||
const boot_handoff_module = b.addModule("boot-handoff", .{
|
||||
.root_source_file = b.path("system/boot-handoff.zig"),
|
||||
});
|
||||
const abi_module = b.addModule("abi", .{
|
||||
.root_source_file = b.path("system/abi.zig"),
|
||||
});
|
||||
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||
// its own module like vfs-protocol — importable by user space, unlike the
|
||||
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||
const device_abi_module = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||
const pci_class_module = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||
// Pure Zig, no kernel imports — one source, two builds.
|
||||
const aml_module = b.addModule("aml", .{
|
||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||
});
|
||||
|
||||
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||
});
|
||||
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
// class requests, descriptors) and the USB class-code taxonomy — the flat
|
||||
// reference the xHCI bus driver, the USB class drivers, and the device
|
||||
// manager's identity matcher all share. Pure data, like pci-class/acpi-ids.
|
||||
const usb_abi_module = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("system/devices/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids_module = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("system/devices/usb-ids.zig"),
|
||||
});
|
||||
// The USB transfer protocol: what a USB class driver says to the xHCI bus
|
||||
// driver to drive its device (open / control / interrupt / bulk). A protocol
|
||||
// module like vfs-protocol, shared by the bus driver and every class driver.
|
||||
const usb_transfer_protocol_module = b.addModule("usb-transfer-protocol", .{
|
||||
.root_source_file = b.path("system/drivers/usb-xhci-bus/usb-transfer-protocol.zig"),
|
||||
});
|
||||
// The block-device protocol: read/write of fixed-size blocks, spoken between a
|
||||
// filesystem and a block driver (usb-storage). A protocol module like the rest.
|
||||
const block_protocol_module = b.addModule("block-protocol", .{
|
||||
.root_source_file = b.path("system/services/block/protocol.zig"),
|
||||
});
|
||||
|
||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||
const parameters_module = b.addModule("parameters", .{
|
||||
.root_source_file = b.path("system/parameters.zig"),
|
||||
});
|
||||
|
||||
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
||||
// The generic kernel imports this as "arch" and never names x86_64, so a new
|
||||
// The generic kernel imports this as "architecture" and never names x86_64, so a new
|
||||
// architecture is a matter of pointing this module at a different directory.
|
||||
const arch_mod = b.addModule("arch", .{
|
||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
||||
const architecture_module = b.addModule("architecture", .{
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // paging uses BootInformation/memory-map + physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // paging works in page_size units
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||
},
|
||||
});
|
||||
// CPU-exception stubs — real assembly, since they need cross-symbol
|
||||
// jumps/calls that Zig inline asm can't express (see the file's header).
|
||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
||||
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/isr.s"));
|
||||
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
||||
// inline asm (it runs relocated to a low page, not at its link address).
|
||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s"));
|
||||
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/trampoline.s"));
|
||||
|
||||
// Firmware-agnostic device discovery. The generic kernel imports this as
|
||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||
// arch module applies to CPU code. The backend is selected at runtime from
|
||||
// the boot handoff (see src/device/platform.zig).
|
||||
const platform_mod = b.addModule("platform", .{
|
||||
.root_source_file = b.path("src/device/platform.zig"),
|
||||
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||
// the boot handoff (see system/devices/platform.zig).
|
||||
const platform_module = b.addModule("platform", .{
|
||||
.root_source_file = b.path("system/devices/platform.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP)
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||
},
|
||||
});
|
||||
|
||||
// Compile-time config the kernel reads as `@import("build_options")`. The
|
||||
// The VFS wire protocol: the vfs sub-project's public interface, exposed as its
|
||||
// own module. Both the vfs server and the runtime's file layer (unistd/stdio)
|
||||
// depend on this contract by name — neither reaches into the other's files. This
|
||||
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||
// expose theirs the same way.
|
||||
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||
});
|
||||
|
||||
// The input wire protocol: the input service's public interface, exposed as its own
|
||||
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||
const input_protocol_module = b.addModule("input-protocol", .{
|
||||
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||
});
|
||||
|
||||
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||
// heap, IPC helpers, the process start shim, device access. This is the stable
|
||||
// application ABI; POSIX compatibility is a separate library on top (see below).
|
||||
// Compiled into every user binary (see addUserBinary), so it inherits each exe's
|
||||
// `.large` code model — do NOT set a target/code_model here. It imports `abi`
|
||||
// for the shared SystemCall numbers / mmap flags, `device-abi` for the device
|
||||
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||
// server. It never touches `boot-handoff` — user space has no business with the
|
||||
// loader↔kernel handoff.
|
||||
const runtime_module = b.addModule("runtime", .{
|
||||
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||
// own module like the other protocol modules. Imported through the runtime.
|
||||
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The USB transfer protocol, so runtime.usb (the class-driver client) can speak
|
||||
// it, the way runtime.input speaks the input protocol.
|
||||
runtime_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||
|
||||
// The display protocol, so runtime.display (the compositor client) and the display
|
||||
// service both speak it through the runtime, like the other protocol modules.
|
||||
const display_protocol_module = b.addModule("display-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||
|
||||
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||
const mmio_module = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||
});
|
||||
|
||||
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||
// (keycode + modifiers -> keysym/character). The `layouts` tables are generated by
|
||||
// tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API over them.
|
||||
// No target set, so each inherits its importer's. See library/xkeyboard-config/.
|
||||
const xkb_layouts_module = b.addModule("layouts", .{
|
||||
.root_source_file = b.path("library/xkeyboard-config/generated/layouts.zig"),
|
||||
});
|
||||
const xkeyboard_config_module = b.addModule("xkeyboard-config", .{
|
||||
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||
});
|
||||
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
const build_options_mod = build_options.createModule();
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||
// The serial-console log sink. Off by default: a real machine often has no
|
||||
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||
|
||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||
@@ -106,65 +469,302 @@ pub fn build(b: *std.Build) void {
|
||||
.abi = .none,
|
||||
});
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/kernel/main.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
|
||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod },
|
||||
.{ .name = "arch", .module = arch_mod },
|
||||
.{ .name = "platform", .module = platform_mod },
|
||||
.{ .name = "build_options", .module = build_options_mod },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// Physical address the bootloader loads the kernel to (identity-mapped under
|
||||
// UEFI). Overrides Zig's default image base so the linker script's layout is
|
||||
// honoured; adjust here if it collides with firmware-reserved memory.
|
||||
exe.image_base = 0x100000; // 1 MiB
|
||||
const kernel_modules = KernelModules{
|
||||
.boot_handoff = boot_handoff_module,
|
||||
.abi = abi_module,
|
||||
.device_abi = device_abi_module,
|
||||
.architecture = architecture_module,
|
||||
.platform = platform_module,
|
||||
.parameters = parameters_module,
|
||||
.initial_ramdisk = initial_ramdisk_module,
|
||||
};
|
||||
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||
|
||||
b.installArtifact(exe);
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||
// then loads these FHS paths off the volume.
|
||||
const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// Boot methods live in src/boot/, one per way of getting the kernel running.
|
||||
// --- init: the first user-space program (a system service) ---
|
||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat is a
|
||||
// serial/test-build diagnostic (the QEMU harness's init tests assert on it, and
|
||||
// -Dserial images emit it), so a flashable image runs a purely event-driven PID 1
|
||||
// that wakes only for real work. The test harness builds with -Dserial=true, so
|
||||
// the heartbeat stays present under test.
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-ids", usb_ids_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||
// the input service. They build chapter-9 class requests from usb-abi.
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb-abi", usb_abi_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
programModule(usb_hid_mouse_exe).addImport("usb-abi", usb_abi_module);
|
||||
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
programModule(usb_storage_exe).addImport("block-protocol", block_protocol_module);
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
// Threaded: the display runs a mouse-listener thread alongside its compositor loop
|
||||
// (docs/threading.md, docs/display.md), so it opts into real atomics/TLS.
|
||||
const display_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||
const shm_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-client", "system/services/shm-client/shm-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
programModule(pci_bus_exe).addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||
// flattened device tree — the aarch64 target flips the default when it
|
||||
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||
// ARM bring-up (fdt).
|
||||
const Discovery = enum { acpi, fdt };
|
||||
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||
const discovery_source: []const u8 = switch (discovery) {
|
||||
.acpi => "system/services/acpi/acpi.zig",
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
programModule(device_manager_exe).addImport("pci-class", pci_class_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
programModule(device_manager_exe).addImport("usb-ids", usb_ids_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||
mk_run.addArg("vfs");
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-mouse");
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-keyboard");
|
||||
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-mouse");
|
||||
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-storage");
|
||||
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||
mk_run.addArg("fat");
|
||||
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||
mk_run.addArg("fat-test");
|
||||
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||
mk_run.addArg("display");
|
||||
mk_run.addFileArg(display_exe.getEmittedBin());
|
||||
mk_run.addArg("display-demo");
|
||||
mk_run.addFileArg(display_demo_exe.getEmittedBin());
|
||||
mk_run.addArg("virtio-gpu");
|
||||
mk_run.addFileArg(virtio_gpu_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-server");
|
||||
mk_run.addFileArg(shm_server_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-client");
|
||||
mk_run.addFileArg(shm_client_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("thread-test");
|
||||
mk_run.addFileArg(thread_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
mk_run.addArg("input");
|
||||
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||
mk_run.addArg("input-source");
|
||||
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||
mk_run.addArg("input-test");
|
||||
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||
mk_run.addArg("args-echo");
|
||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||
mk_run.addArg("process-test");
|
||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||
mk_run.addArg("log-flush");
|
||||
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||
.{ usb_storage_exe, "system/drivers" },
|
||||
.{ fat_exe, "system/services" },
|
||||
.{ display_exe, "system/services" },
|
||||
.{ log_flush_exe, "system/services" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
}
|
||||
|
||||
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||
// which firmware may mirror to a serial console) are silenced by default — a
|
||||
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||
const loader_options = b.addOptions();
|
||||
loader_options.addOption(bool, "serial", serial);
|
||||
const loader_options_module = loader_options.createModule();
|
||||
const efiexe = b.addExecutable(.{
|
||||
.name = "BOOTX64",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/boot/efi.zig"),
|
||||
.root_source_file = b.path("boot/efi.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86_64,
|
||||
.os_tag = .uefi,
|
||||
}),
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod },
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "build_options", .module = loader_options_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
b.installArtifact(efiexe);
|
||||
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
check_fat.addArg("--verify");
|
||||
check_fat.addFileArg(fat_image);
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||
// file then boots every way release media is consumed — flashed raw to a
|
||||
// USB stick with Etcher or dd, or burned to optical media — while
|
||||
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||
mk_iso.addFileArg(fat_image);
|
||||
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||
release_step.dependOn(&iso_install.step);
|
||||
|
||||
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||
// embedded FAT32 image must all agree.
|
||||
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
check_iso.addArg("--verify");
|
||||
check_iso.addFileArg(iso_image);
|
||||
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||
check_iso_step.dependOn(&check_iso.step);
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||
const ovmf_code = b.option(
|
||||
[]const u8,
|
||||
"ovmf-code",
|
||||
"Path to the OVMF_CODE firmware image",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||
@@ -176,7 +776,7 @@ pub fn build(b: *std.Build) void {
|
||||
"ovmf-vars",
|
||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||
@@ -184,15 +784,8 @@ pub fn build(b: *std.Build) void {
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// Assemble an EFI System Partition layout: esp/EFI/BOOT/BOOTX64.efi
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
|
||||
});
|
||||
// The bootloader loads the kernel by name from the volume root, so drop the
|
||||
// kernel ELF at esp/kernel.
|
||||
const kernel_install = b.addInstallArtifact(exe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
||||
});
|
||||
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
@@ -200,6 +793,19 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
const run_efi = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
// "-usb",
|
||||
// "-device",
|
||||
// "usb-ehci,id=ehci",
|
||||
// "-device",
|
||||
// "usb-tablet,bus=usb-bus.0",
|
||||
// "-device",
|
||||
// "usb-mouse,bus=ehci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
@@ -209,10 +815,14 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Present the ESP directory to the guest as a FAT drive.
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-drive",
|
||||
b.fmt("format=raw,file=fat:rw:{s}/esp", .{b.install_path}),
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
@@ -223,14 +833,21 @@ pub fn build(b: *std.Build) void {
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
});
|
||||
// Always capture the guest's serial0 (the kernel's machine-readable log) to a
|
||||
// timestamped file under zig-out, so each run leaves its own log behind.
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ b.install_path, timestamp(b) });
|
||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||
// timestamped file per run.
|
||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
run_efi.step.dependOn(&efi_install.step);
|
||||
run_efi.step.dependOn(&kernel_install.step);
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// const run_cmd = b.addRunArtifact(exe);
|
||||
@@ -243,18 +860,93 @@ pub fn build(b: *std.Build) void {
|
||||
// }
|
||||
|
||||
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
||||
// and can't be executed natively, so only the shared module is unit-tested
|
||||
// here (compiled for the host rather than inheriting a freestanding target).
|
||||
const mod_tests = b.addTest(.{
|
||||
// and can't be executed natively, so only the shared contracts are unit-tested
|
||||
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||
// which also compile-checks that the three-way split stays self-consistent.
|
||||
const test_step = b.step("test", "Run tests");
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||
}) |root| {
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||
}
|
||||
|
||||
// The xkeyboard-config keymap tests need its generated `layouts` import wired, so they
|
||||
// don't fit the plain loop above. Its keycode->character assertions are the end-to-end
|
||||
// proof that the xkb-data -> generator -> Zig-lookup pipeline is correct.
|
||||
const xkb_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
const run_mod_tests = b.addRunArtifact(mod_tests);
|
||||
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||
// loop above.
|
||||
const time_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/time.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||
|
||||
const test_step = b.step("test", "Run tests");
|
||||
test_step.dependOn(&run_mod_tests.step);
|
||||
// runtime.Thread's lock/condvar state machines (Mutex/Condition/RwLock/WaitGroup). Its
|
||||
// Futex seam falls back to std.Thread.Futex off the danos target, so the tests exercise
|
||||
// them with real host threads (docs/threading-plan.md M11). Like time.zig it pulls in
|
||||
// system.zig (syscall wrappers), which needs the `abi` module.
|
||||
const thread_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/thread.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(thread_tests).step);
|
||||
|
||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||
const gen_xkb_step = b.step("gen-xkeyboard-config", "Regenerate library/xkeyboard-config/generated from the vendored data");
|
||||
gen_xkb_step.dependOn(&gen_xkb.step);
|
||||
}
|
||||
|
||||
+180
-15
@@ -35,9 +35,57 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||
blocking (sleep, wait queues) — the leap to a running system.
|
||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||
message-passing channels — the backbone the microkernel's isolated servers will
|
||||
talk over.
|
||||
12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
message-passing channels, then synchronous call/reply between *processes* over
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
16. **[process-management.md](process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
17. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
18. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
19. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
20. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||
21. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -50,9 +98,37 @@ Start with the north star:
|
||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||
fault isolation + live restart — the reincarnation-server + capability model that
|
||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
- **[system-requirements.md](system-requirements.md) — system requirements.** The
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
@@ -64,10 +140,23 @@ Cutting across all of these:
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs.
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||
@@ -94,19 +183,95 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](halting.md)).
|
||||
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
|
||||
## Repository layout
|
||||
|
||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
**A sub-project is addressed by its directory; its entry point repeats the directory's
|
||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||
binary installs to `/system/services/init` (a file at that path), not
|
||||
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||
the identity, the entry file is its implementation. (Same idea as a Go package being its
|
||||
directory, or a macOS `.app` bundle addressed by the bundle, not the executable within.)
|
||||
A sub-project's extra files are reached through the module, never as separate paths.
|
||||
|
||||
```
|
||||
system/ → /system danos's own internals (the self-representation)
|
||||
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||
devices/ the device model /system/devices reflects (+ aml/)
|
||||
device-abi.zig the device wire types (the `device-abi` module)
|
||||
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||
vfs.zig, vfs-test.zig, protocol.zig)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
|
||||
file API (`runtime.fs`) imports by name. `usb`/`block` drivers expose their protocols the
|
||||
same way.
|
||||
|
||||
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||
exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
|
||||
## Source map
|
||||
|
||||
| Area | Code |
|
||||
|------|------|
|
||||
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
|
||||
| Physical frame allocator | `src/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
|
||||
| IPC channels (message passing) | `src/kernel/ipc.zig` |
|
||||
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
||||
| In-kernel test cases | `src/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||
| Big kernel lock + interrupt-safe critical sections | `system/kernel/sync.zig` |
|
||||
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
+61
-7
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
||||
UEFI configuration table
|
||||
│ the loader reads the RSDP's physical address
|
||||
▼
|
||||
BootInfo.acpi_rsdp (u64, in the shared `danos` module) src/root.zig
|
||||
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInfo
|
||||
▼
|
||||
platform.discover(boot_info, …) src/device/platform.zig
|
||||
platform.discover(boot_info, …) system/devices/platform.zig
|
||||
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
||||
▼
|
||||
acpi.discover(rsdp_phys, …) src/device/acpi.zig
|
||||
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||
│ dereferences the RSDP, reads the pointer it contains
|
||||
▼
|
||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||
@@ -35,7 +35,7 @@ successor the **XSDT** (ACPI 2.0+) — which in turn lists every other SDT.
|
||||
## Step 1 — the loader finds the RSDP
|
||||
|
||||
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
||||
still up. `acpiRootSystemDescriptorPointer()` in `src/boot/efi.zig` walks the UEFI
|
||||
still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
|
||||
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
||||
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
||||
the [memory map](memory-map.md).
|
||||
@@ -44,11 +44,11 @@ the [memory map](memory-map.md).
|
||||
|
||||
The loader can't just call the device module: the bootloader binary and the kernel
|
||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||
module at all** (it imports only the shared `danos` module). So instead of a call, it
|
||||
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||
deposits a value in the handoff struct:
|
||||
|
||||
```zig
|
||||
// src/boot/efi.zig — while boot services are still up
|
||||
// boot/efi.zig — while boot services are still up
|
||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||
```
|
||||
|
||||
@@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a
|
||||
part that answers "where are the tables?" — everything past the RSDP is just following
|
||||
more pointers the tables themselves provide.
|
||||
|
||||
## ACPI events: the SCI, the power button, and GPEs (M21)
|
||||
|
||||
The tables above are static description; ACPI is also a *live* channel. Hardware
|
||||
raises the **SCI** (System Control Interrupt) — one shared, level-triggered line
|
||||
whose vector the FADT names — and the OS reads status registers to learn what
|
||||
happened: a fixed event like the power button, or a **General-Purpose Event**
|
||||
(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML
|
||||
to ring 3, the event side lives there too, in the same **acpi service** — the
|
||||
device discoverer and the event source are one process, because both need the
|
||||
namespace and the port grant.
|
||||
|
||||
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||
header, while the blob resources are header-stripped bytecode that starts with
|
||||
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||
finds the line to `irq_bind`.
|
||||
|
||||
With those in hand the service enables ACPI mode (only if `SCI_EN` is clear —
|
||||
some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||
|
||||
- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status.
|
||||
The handler clears it (write-1-to-clear), logs the press, and publishes a
|
||||
[`power`](power.md) `power_button` event to subscribers.
|
||||
- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the
|
||||
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||
method, drains the **Notify** queue that method produced, maps each notified
|
||||
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||
per evaluation; everything else a handler needs (field access, control flow,
|
||||
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||
|
||||
**How this is tested.** QEMU cannot raise GPEs deterministically on this config,
|
||||
so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||
are interface-complete but validated on real hardware later.
|
||||
|
||||
The service surface these events are *published on* — subscription, the event
|
||||
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||
|
||||
## Related
|
||||
|
||||
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
||||
the ACPI-reclaim memory the RSDP lives in.
|
||||
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
||||
tables into one neutral device model shared with the ARM device-tree path.
|
||||
tables into one neutral device model shared with the ARM device-tree path, and how
|
||||
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||
module and never names ACPI directly.
|
||||
|
||||
+13
-13
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
||||
|
||||
```zig
|
||||
const arch_mod = b.addModule("arch", .{
|
||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
});
|
||||
```
|
||||
|
||||
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
|
||||
```
|
||||
|
||||
Adding a second architecture is then a build-time choice: create
|
||||
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||
`system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
||||
boundary _is_ the architecture interface** — when a new arch is missing a function
|
||||
the generic kernel calls, the build fails and names exactly what's missing.
|
||||
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
|
||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||
register, or a memory-management structure? If so, it's arch-specific.
|
||||
|
||||
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core |
|
||||
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||
|---|---|
|
||||
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
||||
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
||||
@@ -51,9 +51,9 @@ should end up on the generic side; the arch module stays small.
|
||||
There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||
`src/kernel/arch/<cpu>/`.
|
||||
`system/kernel/arch/<cpu>/`.
|
||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds
|
||||
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||
separate loader at all — the firmware jumps straight into the kernel with a
|
||||
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
||||
@@ -61,29 +61,29 @@ There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
## Current x86_64 contents
|
||||
|
||||
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
||||
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
||||
trap frame.
|
||||
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
|
||||
[paging.md](paging.md)).
|
||||
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
||||
port-I/O + MSR primitives.
|
||||
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
||||
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
||||
express them.
|
||||
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||
address, one PT_LOAD per permission set).
|
||||
|
||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||
x86_64 [SysV](sysv.md), via the shared `danos.kernel_abi`), but it's three lines
|
||||
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||
|
||||
+4
-4
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
|
||||
new ISA.
|
||||
|
||||
They are as different from each other as either is from x86-64: separate registers,
|
||||
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`.
|
||||
page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
|
||||
|
||||
## The Raspberry Pi models
|
||||
|
||||
@@ -57,16 +57,16 @@ the DTB/ACPI tells you what devices exist.
|
||||
|
||||
## What danos needs, layer by layer
|
||||
|
||||
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||
- **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||
providing the same `arch` interface as x86_64: `halt`, context switch,
|
||||
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is
|
||||
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
|
||||
planned (see the decision above), so there's a single ARM backend to write.
|
||||
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
||||
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
||||
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
||||
from a different source. This is where keeping boot-protocol knowledge on the
|
||||
loader side (as we did for the UEFI memory-map classification) pays off.
|
||||
- **The UEFI loader mostly carries over.** `src/boot/efi.zig` is largely
|
||||
- **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
|
||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
||||
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
||||
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
# Coding standards
|
||||
|
||||
Conventions for danos source. The overriding one, from which most of the rest follows:
|
||||
|
||||
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||
> abbreviation is an acronym.**
|
||||
|
||||
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||
that trade is not close.
|
||||
|
||||
## The rule, precisely
|
||||
|
||||
**Acronyms and initialisms stay.** They *are* the full name — expanding them would make
|
||||
the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`,
|
||||
`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`,
|
||||
`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`,
|
||||
`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`,
|
||||
`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention
|
||||
demands: `Hal` the type, `hal` the variable, `mapMmio` the function.
|
||||
|
||||
**Everything else is spelled out.** If it's a word with letters removed, restore them:
|
||||
|
||||
| Abbreviation | Full |
|
||||
|---|---|
|
||||
| `proto` | `protocol` |
|
||||
| `msg` | `message` |
|
||||
| `desc` | `descriptor` |
|
||||
| `res` | `resource` |
|
||||
| `recv` | `receive` |
|
||||
| `buf` | `buffer` |
|
||||
| `cur` | `current` |
|
||||
| `src` / `dst` | `source` / `destination` |
|
||||
| `idx` | `index` |
|
||||
| `addr` | `address` |
|
||||
| `reg` | `register` |
|
||||
| `prev` | `previous` |
|
||||
| `cfg` / `config` | `configuration` |
|
||||
| `arch` | `architecture` |
|
||||
| `sched` | `scheduler` |
|
||||
| `dev` | `device` |
|
||||
| `sys` / `syscall` | `system` / `system_call` |
|
||||
| `info` | `information` |
|
||||
| `dt` | `device_tree` |
|
||||
| `ep` | `endpoint` |
|
||||
| `rt` | `runtime` |
|
||||
| `func` | `function` |
|
||||
| `phys` / `virt` | `physical` / `virtual` |
|
||||
| `wq` | `wait_queue` |
|
||||
|
||||
This list is illustrative, not exhaustive. The rule is the rule; when you meet a new
|
||||
abbreviation, expand it.
|
||||
|
||||
## Exceptions
|
||||
|
||||
Three, and only three.
|
||||
|
||||
1. **Foreign ABI names are spelled exactly as the ABI spells them — but only inside
|
||||
the layer that *is* that ABI.** A function that *is* the C or POSIX interface keeps
|
||||
its name: `fopen`, `fwrite`, `fread`, `malloc`, `calloc`, `realloc`, `free`,
|
||||
`memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, `close`, `lseek`, `stat`,
|
||||
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||
`fwrite` any more.
|
||||
|
||||
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||
layer is premature until danos actually needs it (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||
`system` interface, so it keeps `open`/`read`/`errno`/`O_CREAT`. **Everywhere else,
|
||||
Zig/danos naming applies with no exception**: a concept POSIX also has gets a danos
|
||||
name — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||
flag, not `O_CREAT`; the boundary is where `stat`→`status` and `O_CREAT`→`create` get
|
||||
mapped. (The `syscall` *wrappers* elsewhere are not an exception — they wrap the
|
||||
private danos ABI, so they use danos names.)
|
||||
|
||||
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||
not ours, and are left alone:
|
||||
- **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a
|
||||
shortening of "initialize".
|
||||
- **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own
|
||||
structs use bare `len`/`ptr` fields to mirror them, so a reader carries one
|
||||
mental model. (Compounds still expand: a field is `message_len`, not
|
||||
`message_length` — `len` is kept, `msg` is not.)
|
||||
- The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are
|
||||
Zig's spelling.
|
||||
|
||||
The rule governs the names *we* coin.
|
||||
|
||||
3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may
|
||||
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||
|
||||
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
## A note on collisions
|
||||
|
||||
Two identifiers can legitimately expand to the same word. When they do, keep both
|
||||
meaningful by renaming one to its *specific* identity rather than the generic
|
||||
expansion. Two cases resolved this way:
|
||||
|
||||
- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would
|
||||
collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module
|
||||
became **`parameters`**, which is what it holds.
|
||||
- The kernel `device.zig` module would collide with `dev` (a device value) at
|
||||
`device`. The module alias became **`device_model`**, which is what it is — the
|
||||
device data model (`Device`, `DeviceTree`, `ResourceKind`).
|
||||
- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a
|
||||
`Namespace` **instance**. Resolved by dropping the module alias entirely — the two
|
||||
types it provided are imported directly (`const Node = @import("namespace.zig").Node;`)
|
||||
— which frees `namespace` for the instance.
|
||||
|
||||
A related case is one abbreviation with two meanings. In the AML code, `op` means
|
||||
**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` /
|
||||
`LogicOperation` means **operation** — distinguished by case. The per-opcode parser
|
||||
handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the
|
||||
opcode's structure, which says what they do without overloading "op".
|
||||
|
||||
## Case and file names
|
||||
|
||||
Within those spelling rules, follow Zig's own conventions:
|
||||
|
||||
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||
- **Variables, fields, constants** — `snake_case`: `message_length`, `devices_broker`,
|
||||
`notify_badge_bit`.
|
||||
|
||||
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
|
||||
## Named values, not magic numbers
|
||||
|
||||
The naming rule has a twin: **a value with meaning gets a name, too.** The same
|
||||
principle drives both — a reader should never have to leave the code to understand it.
|
||||
An abbreviated *name* forces a reader to guess; a bare *number* forces them worse, out
|
||||
to a spec or a header or a comment three files away, to learn what the value even *is*.
|
||||
If `0x0C` is the PCI serial-bus class, the code says `BaseClass.serial_bus`, not `0x0C`;
|
||||
if `0x04` is the ACPI IRQ resource descriptor, it says `SmallResourceType.irq`, not
|
||||
`0x04`. The number is an implementation detail of the name — recorded once, where the
|
||||
name is defined, and never spelled again at a use site.
|
||||
|
||||
**Prefer an `enum`** when the values form a set (device classes, AML opcodes, resource
|
||||
descriptor types, states): the type then also says *which* set a value belongs to, and
|
||||
the compiler rejects a value from the wrong one. A lone `pub const` with a descriptive
|
||||
name suffices for a one-off (`const large_descriptor_bit = 0x80`). Reach for the enum
|
||||
the moment code elsewhere compares against, packs, or produces the value — a packed PCI
|
||||
class triple is written from named parts (`.serial_bus`, `.usb`, `.xhci`), never as
|
||||
`0x0C_03_30` under a comment that decodes the bytes.
|
||||
|
||||
The exceptions are the numbers that carry no hidden meaning: `0` and `1` as plain zero
|
||||
and one, an index step, a field width, a bit shift. `x + 1`, `buffer[0]`, and `<< 8`
|
||||
need no christening — there is nothing to look up. The test is exactly the naming test:
|
||||
*would a reader have to look this up to know what it means?* If yes, name it. This is
|
||||
what `opcodes.zig`'s `*_opcode` constants, `acpi-ids`'s `HardwareId`, and `pci-class`'s
|
||||
class enums already are — reference data defined once and named everywhere it is used.
|
||||
|
||||
## Why acronyms are the line
|
||||
|
||||
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||
input output" in code — that expansion is what the acronym *is for*. But `msg` is just
|
||||
`message` with three letters stolen, and stealing them buys nothing a reader wants. The
|
||||
test for "is this an abbreviation I must expand" is simply: *is there a longer word this
|
||||
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||
phrase, leave it.
|
||||
|
||||
## Zen of Zig
|
||||
|
||||
* Communicate intent precisely.
|
||||
* Edge cases matter.
|
||||
* Favor reading code over writing code.
|
||||
* Only one obvious way to do things.
|
||||
* Runtime crashes are better than bugs.
|
||||
* Compile errors are better than runtime crashes.
|
||||
* Incremental improvements.
|
||||
* Avoid local maximums.
|
||||
* Reduce the amount one must remember.
|
||||
* Focus on code rather than style.
|
||||
* Resource allocation may fail; resource deallocation must succeed.
|
||||
* Memory is a resource.
|
||||
* Together we serve the users.
|
||||
@@ -0,0 +1,121 @@
|
||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||
|
||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
|
||||
|
||||
## Directory structure
|
||||
|
||||
| Path | Description |
|
||||
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||
| /etc | Host-specific system-wide configuration files. |
|
||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||
| /sbin | Essential system binaries (e.g init) |
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||
|
||||
## File types
|
||||
|
||||
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||
|
||||
| type | symbol | Description |
|
||||
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||
|
||||
## /dev
|
||||
|
||||
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. This is what
|
||||
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
|
||||
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
|
||||
directories at all and `kind` hardcoded to zero. The three sections below describe the
|
||||
intended shape, and are honest about which parts the kernel can already support.
|
||||
|
||||
### Character devices
|
||||
|
||||
A character device is a byte stream with no addressable position: bytes are delivered
|
||||
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||
is already that program, minus the file-node client half.
|
||||
|
||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||
easiest first entry.
|
||||
|
||||
### Block devices
|
||||
|
||||
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||
and other persistent storage are the whole population of this class.
|
||||
|
||||
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||
|
||||
### Pseudo-devices
|
||||
|
||||
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||
yielding unpredictable bytes.
|
||||
|
||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||
sensible place to start, because they are exactly the entries that need no driver
|
||||
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
|
||||
out of its own address space — `null` and `zero` are a few lines each in
|
||||
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
|
||||
structure that every later device node depends on and that the flat ramfs currently
|
||||
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
|
||||
`Stat.kind`, so that a caller can tell a character device from a regular file.
|
||||
|
||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||
it until it is a real one.
|
||||
+68
-12
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
|
||||
PIC. There are two halves; we only need one so far:
|
||||
|
||||
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
||||
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`.
|
||||
CPU's own timer and receives interrupts routed to it. `system/kernel/architecture/x86_64/apic.zig`.
|
||||
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
||||
Not needed for the timer — it'll arrive with the keyboard.
|
||||
|
||||
@@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea
|
||||
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
||||
timer — a later step.
|
||||
|
||||
### Is the TSC trustworthy? Invariant, and synchronized
|
||||
|
||||
A cycle counter is only a valid *clock* if two things hold, and danos checks both,
|
||||
because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET.
|
||||
|
||||
**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with
|
||||
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||
measured frequency without the invariance guarantee is not enough.
|
||||
|
||||
**Synchronized.** Each core has its own TSC. Even invariant ones can start at different
|
||||
values (a second socket, some firmware), so a thread migrating from a core reading
|
||||
`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor
|
||||
drift with frequency. It costs a memory-mapped read instead of a register read, but it
|
||||
keeps time *accurate*, which is the whole point. The switch preserves the current value,
|
||||
so the clock never jumps. The boot log names the outcome:
|
||||
|
||||
```
|
||||
/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD
|
||||
/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG)
|
||||
```
|
||||
|
||||
## Two kinds of vector, one dispatch
|
||||
|
||||
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
||||
@@ -89,7 +123,6 @@ if (state.vector < 32) {
|
||||
on_fault(state); // exception: report and halt (never returns)
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
handler(); // device: run the registered handler
|
||||
apic.eoi(); // ...acknowledge the LAPIC
|
||||
}
|
||||
// else: spurious/unhandled — deliberately no EOI
|
||||
```
|
||||
@@ -100,10 +133,23 @@ Two things make device interrupts *return* where exceptions don't:
|
||||
flows back to `isr_common`, which restores every register it saved and executes
|
||||
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
||||
saves *all* the general registers.)
|
||||
2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss
|
||||
2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss
|
||||
this and the LAPIC thinks we're still busy and never delivers the next
|
||||
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
||||
|
||||
**Each handler issues its own EOI**, rather than the dispatcher doing it around the
|
||||
call. That looks like a needless devolution while the timer is the only device, and
|
||||
`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early,
|
||||
because the tick hook is the scheduler, which may switch tasks and not return
|
||||
promptly — the LAPIC mustn't wait on it).
|
||||
|
||||
It stops looking needless with the second device. A *routed* interrupt — one arriving
|
||||
through the I/O APIC from a real device line — must be **masked before it is
|
||||
acknowledged**, because a level-triggered line is still asserted at EOI time and would
|
||||
redeliver instantly, forever. Only the handler knows which discipline its source
|
||||
needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the
|
||||
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||
|
||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||
a handler must not use them; ours don't.)
|
||||
@@ -131,14 +177,24 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the
|
||||
count would stay put and the test would fail. That it advances — while the CPU was
|
||||
spinning in unrelated code — is the whole mechanism working end to end.
|
||||
|
||||
## Since (done elsewhere)
|
||||
|
||||
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||
[drivers.md](drivers.md).
|
||||
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||
user drivers — see [paging.md](paging.md).
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read
|
||||
scancodes from the PS/2 controller — the first *input* device.
|
||||
- **`sleep()` / timeouts** built on the calibrated clock (the monotonic
|
||||
`uptimeMs()` is in place).
|
||||
- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like
|
||||
the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO
|
||||
marked uncacheable (via the page's cache bits or an MTRR).
|
||||
- **Preemption**: once there are tasks, the timer handler is where the scheduler
|
||||
decides to switch — the reason a *returning* interrupt matters.
|
||||
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and port I/O is
|
||||
now available to ring 3 via the claim-gated `io_read`/`io_write` syscalls
|
||||
([drivers.md](drivers.md)) — so the first *input* device is unblocked; it just needs
|
||||
writing (claim the controller, `irq_bind` GSI 1, read scancodes from `0x60`).
|
||||
- **MSI-X**: `msi_bind` gives one per-device edge-triggered vector (M15); MSI-X's
|
||||
multi-vector table (many queues per device, e.g. NVMe) is the remaining extension.
|
||||
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||
|
||||
@@ -0,0 +1,172 @@
|
||||
# The device manager
|
||||
|
||||
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||
root-hub ports and reports each connected device (`child_added`); the manager
|
||||
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||
for drivers.
|
||||
|
||||
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||
document: that is the universal lifecycle every danos process speaks —
|
||||
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||
`runtime.process` interface. The device manager is that design's first serious
|
||||
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||
driver is stopped, health-checked, and buried exactly like any other process.
|
||||
|
||||
## The tree: structure in the manager, authority in the kernel
|
||||
|
||||
The device tree is two things fused: *information* (what exists, how it nests) and
|
||||
*authority* (a descriptor is a licence to map physical memory). They separate:
|
||||
|
||||
- The **kernel keeps the capability system** — device, I/O-port, and interrupt
|
||||
claims, resource containment on `device_register`, the
|
||||
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||
dies** (settled; it is increment 1 of
|
||||
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||
buggy one would un-earn everything the microkernel bought.
|
||||
- The **device manager owns the tree as data** — identity, topology, naming, driver
|
||||
matching, hotplug events, and being the one process everything else asks about
|
||||
devices. Firmware discovery seeds it (today via the kernel's snapshot); **bus
|
||||
drivers grow it** by reporting what they see; applications query and watch it.
|
||||
`device_enumerate` fades to a manager-internal (then deleted) seam.
|
||||
|
||||
Long-term, discovery itself leaves the kernel — but not *into* the manager. PCI
|
||||
enumeration is a **pci-bus driver**: the manager spawns it against the host bridge
|
||||
(already a device with the ECAM window as a resource), it scans, it reports functions
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
instead of appending a duplicate. The kernel table has no unregister, so without
|
||||
this a restarted registering bus would re-report its children as fresh nodes on
|
||||
every respawn. Idempotence is what makes restart-and-re-report sound for *every*
|
||||
reporting bus — pci-bus, the acpi service, a future fdt service — not just one,
|
||||
and it is why supervision (below) can prune a dead bus's subtree and trust the
|
||||
restarted instance to rebuild exactly the same ids.
|
||||
|
||||
## The protocol
|
||||
|
||||
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||
one world.
|
||||
|
||||
| Direction | Message | Purpose |
|
||||
|---|---|---|
|
||||
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||
| app → manager | `subscribe` | receive published add/remove events |
|
||||
|
||||
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||
On a death notification:
|
||||
|
||||
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||
input service losing, then regaining, a keyboard is the *honest* description of
|
||||
what happened. The restarted instance rediscovers and re-reports.
|
||||
3. **The claim is already free** because the kernel released it at death — the
|
||||
restarted instance claims the same controller and comes up.
|
||||
|
||||
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||
services it starts. If the manager dies, drivers keep running (they hold their
|
||||
claims; the kernel doesn't care who their supervisor was — though their exit
|
||||
notifications now dangle harmlessly). The restarted manager re-learns the world:
|
||||
kernel snapshot, then a re-`hello` round — drivers answer a broadcast or are stopped
|
||||
and respawned. Full state handoff is deliberately not attempted.
|
||||
|
||||
## Thin drivers, class protocols
|
||||
|
||||
The [driver-model.md](driver-model.md) three-shape split, restated as processes:
|
||||
|
||||
- A **bus driver** (usb-xhci-bus) owns its controller — claim, MMIO, IRQ/MSI, DMA
|
||||
rings — and offers a *transfer* protocol ("submit a control transfer to device N",
|
||||
built from the usb-abi request constructors) plus tree reports to the manager.
|
||||
- A **class driver** (usb-hid, usb-storage) owns nothing: it is matched to a reported
|
||||
child by its identity triple, speaks the bus's transfer protocol downward and its
|
||||
service's protocol upward — HID reports to the input service, blocks to the block
|
||||
service. It works unchanged over any controller.
|
||||
- **Services** (input, display, block) aggregate class drivers and face applications.
|
||||
|
||||
Each arrow is a protocol module. The manager routes none of the data plane — it
|
||||
introduces the parties (matching), supervises them (lifecycle), and gets out of the
|
||||
way.
|
||||
|
||||
## Increments
|
||||
|
||||
Increments 1–4 are the lifecycle prerequisites and live in
|
||||
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `runtime.process`). On top of those:
|
||||
|
||||
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||
usb-xhci-bus becomes the first conforming driver.
|
||||
6. **Tree reports**: `child_added`/`child_removed`; the manager mirrors; xHCI reports
|
||||
the mouse and keyboard QEMU already hangs off it.
|
||||
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||
PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3
|
||||
— each in a single phase so no device is ever matched from both sources at
|
||||
once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus
|
||||
already reports PCI functions (M20.2).
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
- **Stateful buses**: pruning the subtree on bus-driver death is right for USB. A
|
||||
future storage bus with in-flight writes wants drain-before-terminate — which is
|
||||
exactly the `deadline_ms` parameter `stop()` already has; a per-driver deadline
|
||||
is one value in the manager's policy table when such a bus arrives. No design
|
||||
change.
|
||||
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||
the world (above). Checkpointing driver state with the manager is deferred until
|
||||
something demonstrates the need.
|
||||
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||
@@ -167,3 +167,81 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
|
||||
## Update (M20.3, 2026-07-13): ACPI enumeration left the kernel too
|
||||
|
||||
The kernel no longer folds the AML namespace's Device objects into the device
|
||||
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||
entirely in user space; the kernel seeds only the host bridge and the
|
||||
acpi-tables node.
|
||||
|
||||
## Discovery is a swappable process per firmware (M19–M20)
|
||||
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
|
||||
- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi
|
||||
service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML.
|
||||
- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is
|
||||
an **fdt service**: it claims a `devicetree-blob` node and walks the tree —
|
||||
pure data, no bytecode, so it needs neither a port grant nor an interpreter,
|
||||
strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md)
|
||||
bring-up fills it in.)
|
||||
|
||||
The device manager spawns the discoverer under the **neutral ramdisk name
|
||||
`discovery`** and never learns which firmware it is on; the build's
|
||||
`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the
|
||||
aarch64 target flips the default when it lands). The manager owns the device
|
||||
tree as *data* and touches no hardware, ever — firmware bytecode runs only
|
||||
inside the crashable, supervised discoverer, so an AML fault can never take
|
||||
down the supervisor.
|
||||
|
||||
Two consequences of neutrality bind on later work:
|
||||
|
||||
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||
`ServiceId.power`, and subscribers never learn the difference.
|
||||
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||
report a real node.
|
||||
|
||||
Two supporting decisions keep the kernel's remaining slice honest:
|
||||
|
||||
- **The AML interpreter is a shared build module**, compiled into both the
|
||||
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||
device count across the ring-3 move.
|
||||
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||
PCI functions carry BAR resources, and `device_register` containment demands
|
||||
the bridge own windows that cover them. Those apertures are derived
|
||||
kernel-side from the boot memory map's MMIO holes (regions that are neither
|
||||
RAM nor tables) — mechanical, AML-free, and available at boot regardless of
|
||||
what later moved to user space. The acpi service's authority is likewise
|
||||
exactly one node: the `acpi-tables` node, whose broad io_port grant is the
|
||||
documented trust boundary for the one process allowed to run firmware
|
||||
bytecode.
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
# Display service — build plan (v1: the dumb-framebuffer compositor)
|
||||
|
||||
The ordered, checkpointable build-out for [display.md](display.md). Each milestone is
|
||||
small, lands on its own, and ends in a **verifiable gate** — shaped for a `/loop` run.
|
||||
Read [display.md](display.md) first for the *why*; this is the *what* and the *order*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **Handoff = device node + write-combining `mmio_map`.** The kernel seeds a synthetic
|
||||
`display0` node from `BootInformation.framebuffer`; the service claims + WC-maps it.
|
||||
(Not a bespoke `framebuffer_map` syscall — the device route inherits ownership,
|
||||
release-on-death, and re-claim-on-restart.)
|
||||
- **v1 = the full compositor pipeline on the dumb framebuffer.** One `display` service
|
||||
owns the LFB + a cacheable back buffer + a layer stack; double-buffer + damage-driven
|
||||
present; clients draw via server-side commands. **No** runtime mode-setting, **no**
|
||||
shared-memory surfaces — both deferred (see display.md, "What v1 does not do").
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||
pitch/format blits are all host-testable with a fake framebuffer).
|
||||
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||
|
||||
---
|
||||
|
||||
## D1 — The handoff primitive (kernel) ✅
|
||||
|
||||
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||
|
||||
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||
|
||||
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||
screenshot-of-a-fill gate because it proves the *actual* WC property headlessly; the
|
||||
visible fill folds into D2's gate (the service clears the screen through the back buffer).
|
||||
Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `device-list`,
|
||||
`device-manager` all still pass with the +1 device in the table.
|
||||
|
||||
## D2 — Service skeleton, protocol, runtime module ✅
|
||||
|
||||
Stand up the named service and the double-buffer, no layers yet.
|
||||
|
||||
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||
until D3. Init clears the back buffer and presents it — the double-buffer path.
|
||||
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||
lookup with retry (model: block.zig).
|
||||
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||
`/system/services/display`.
|
||||
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||
a fixed kernel-stack `frames` array. Rewrote `systemMmap` to map page-by-page with
|
||||
rollback (no scratch array) and raised the cap to 8192 pages (32 MiB) — enough for a
|
||||
4K back buffer. A real limitation met, exactly the kind this project chases.
|
||||
|
||||
**Gate (met, automated):** `python3 test/qemu_test.py display-service` spawns the
|
||||
compositor and matches its own serial heartbeats — `display: online {w}x{h} pitch …`
|
||||
followed by `display: presented frame 0` — which it prints only after the whole
|
||||
claim → WC-map → back-buffer → clear → present chain succeeds (matched on serial like the
|
||||
fault cases, since a lone blocking service can't reschedule the in-kernel test context to
|
||||
poll). Regression-checked: `usermem`, `heap` (the `mmap` rewrite), `init` (the boot-list
|
||||
addition), and D1's `display` all still pass.
|
||||
|
||||
## D3 — Layer stack + compositor + damage present ✅
|
||||
|
||||
The heart: composite an ordered layer stack, present only what changed.
|
||||
|
||||
- [x] A layer table (16 slots): each `Layer` = position, z, visible, a server-owned
|
||||
`mmap`'d surface (freed on `destroy_layer`). `damage` accumulates the dirty screen
|
||||
region since the last present.
|
||||
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||
`damage`, `present`.
|
||||
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||
Colour packing (rgbx/bgrx) is `protocol.pack`, also host-tested.
|
||||
- [x] Host tests (`zig build test`, green): rect intersect/unite, `fillRect` clipping +
|
||||
`stride > width` padding, `composite` overlap-shows-top + damage clipping, `blitTile`
|
||||
unaligned read + clipping, and `pack` for both pixel formats.
|
||||
|
||||
**Gate (met):** `zig build test` green for the compositor + pack unit tests, **and** the
|
||||
`display-service` case's startup self-check composites two overlapping layers on the real
|
||||
framebuffer and reads back the composited pixels — overlap = top layer, outside = bottom
|
||||
layer — logging `display: compositor self-check ok` (matched by the harness).
|
||||
|
||||
## D4 — Client API + the demo client ✅
|
||||
|
||||
Prove the pipeline end-to-end from a separate process.
|
||||
|
||||
- [x] Finished [runtime/display.zig](../library/runtime/runtime.zig): a `Layer` handle with
|
||||
`fill` / `blitTile` (inline tile) / `configure` (move/restack/show) / `damage` /
|
||||
`destroy`, `createLayer`, and a `color(r,g,b)` helper (caches the mode, packs via
|
||||
`protocol.pack`). Coordinates are signed over the wire (`@bitCast` both ways).
|
||||
- [x] `system/services/display-demo/`: a hardware-free client (the `input-source` analog)
|
||||
— a full-screen wallpaper layer, a rectangle that slides back and forth (moved by
|
||||
`configure` each frame, so the compositor repaints old + new), and a cursor layer;
|
||||
presents in a loop paced by `runtime.time`. Wired into build + initial-ramdisk.
|
||||
- [x] **Bug this surfaced:** `protocol.message_maximum` was 4096, but the kernel caps
|
||||
every IPC message at `MESSAGE_MAXIMUM` = 256 — so `replyWait` rejected the oversized
|
||||
receive buffer with `-E2BIG` and the serve loop had been *spinning* since D2 (unseen,
|
||||
as D2/D3 matched init-time heartbeats). Set it to 256; `blit_tile` is now explicitly
|
||||
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shm surface.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display-demo` spawns the service + `display-demo`;
|
||||
the demo drives a run of frames of motion through the layer client API and logs
|
||||
`display-demo: ok` (the visible motion is a screenshot via `zig build run-x86-64`).
|
||||
Regression-checked: `zig build test`, `display` (D1), and `display-service` (D2/D3) all
|
||||
still pass, and the default `zig build` is clean.
|
||||
|
||||
## D5 — Test cases + docs ✅
|
||||
|
||||
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||
the pure host tests (`zig build test`).
|
||||
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||
memory marked DONE with the commits.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||
`zig build test` is green, and the default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## v1 status: complete
|
||||
|
||||
D1–D5 done. The display service is a working framebuffer compositor: it owns the
|
||||
framebuffer (write-combining), composites a z-ordered layer stack into a cacheable back
|
||||
buffer, presents only the damaged region, and is driven over IPC by the `runtime.display`
|
||||
client — proven end-to-end by a separate demo process. Two limitations are deliberate and
|
||||
documented (docs/display.md): no runtime mode-setting (native backend) and no true vsync
|
||||
(no vblank on a dumb framebuffer). Next steps are the Deferred items below.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Shared-memory surfaces** — generalize M13 capability passing to memory objects
|
||||
(`shm_create`/`shm_map`), so bitmap clients hand the compositor a rendered surface
|
||||
instead of drawing commands. The compositor's layer model already anticipates it.
|
||||
- **Native backend (Bochs DISPI, then virtio-gpu)** — behind the same internal backend
|
||||
interface as the dumb framebuffer: EDID mode list + runtime resolution/bpp change +
|
||||
(eventually) a vblank/flip path for true vsync.
|
||||
- **Driver/compositor process split** — only when a second backend or a second head makes
|
||||
the abstraction pay for itself.
|
||||
@@ -0,0 +1,173 @@
|
||||
# Display v2 — build plan (pluggable scanout: GOP floor + virtio-gpu native)
|
||||
|
||||
The ordered, checkpointable build-out for [display-v2.md](display-v2.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-plan.md](display-plan.md). Read display-v2.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + present/flush + vsync).
|
||||
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||
ever," not a live fall-back after a reprogram.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shm-backed) and the back buffer
|
||||
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||
"it's on screen."
|
||||
|
||||
- `zig build test` — host unit tests (backend selection, virtio struct sizes/encodings,
|
||||
pixel-check helpers).
|
||||
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial markers.
|
||||
The virtio cases boot with `-device virtio-gpu` (a per-case `qemu_extra`).
|
||||
- `run-x86-64` renders to a window — for the human's own satisfaction, **not** a gate.
|
||||
|
||||
---
|
||||
|
||||
## V1 — The scanout backend seam (refactor, no behaviour change) ✅
|
||||
|
||||
Extract scanout from the compositor so today's path becomes one backend among future ones.
|
||||
|
||||
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||
(`canModeSet`/`hasVsync`, both false for GOP).
|
||||
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||
framebuffer geometry left in the compositor core.
|
||||
- [x] The selection decision is the pure `chooseKind(native_available)` (gop unless a
|
||||
native driver announced), split from the syscall-bound `select()`/`Gop.init()`.
|
||||
|
||||
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||
is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The `shm` cross-process memory capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shm arena → returns virtual_address + handle; `shm_map(cap)`
|
||||
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||
dispatch by kind, so an `ShmObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||
frames — the object owns them.
|
||||
- [x] `library/runtime/shm.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
`map(handle) -> ptr`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py shm` — `shm-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shm-server` as an `ipc_call` send_cap; the server
|
||||
`shm_map`s it and reads the **same bytes** back → `shm: shared 4096 bytes ok`. Guardrail:
|
||||
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||
tests all still pass — the handle-table change broke no existing IPC.
|
||||
|
||||
## V3 — The virtio-gpu driver: bring-up + a frame on screen ✅
|
||||
|
||||
- [x] `system/drivers/virtio-gpu/`: claim the virtio-gpu PCI function (device-manager
|
||||
match on the display/other class triple, driver self-confirms vendor 0x1AF4/device
|
||||
0x1050 from config space), enable memory-space + bus-master, walk the vendor
|
||||
capabilities in config space to find common-config + notify, map the BAR, negotiate
|
||||
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||
shm-shared surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||
|
||||
**Gate (met):** the `virtio-gpu` case (QEMU `-device virtio-gpu-pci`) boots the
|
||||
device-manager stack, which discovers the function and spawns the driver; the driver writes
|
||||
a known test pattern into the scanout backing, `transfer_to_host_2d` + `resource_flush`es
|
||||
it, and **waits for the device's used-ring ack**, then reads the backing back and checks the
|
||||
pattern — logging `virtio-gpu: scanout 640x480 online` and `virtio-gpu: flush acked, pixel
|
||||
check ok`. That proves virtqueue + resource + attach + set_scanout + transfer + flush end to
|
||||
end without a screenshot (the used-ring ack is the device confirming it consumed the frame).
|
||||
|
||||
## V4 — The native backend + hot-attach ✅
|
||||
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared `shm` scanout surface
|
||||
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||
- [x] The driver **announces** to `.display` after bring-up (looks it up with a bounded retry,
|
||||
sends `attach_scanout` with the geometry + the shared surface as an `ipc_call` send_cap).
|
||||
The compositor maps it, looks up `.scanout` itself (no need to pass the endpoint — the
|
||||
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shm_physical` (a
|
||||
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||
|
||||
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||
boots the whole system) starts the compositor + `display-demo` + device-manager; the driver
|
||||
announces, the compositor logs `display: scanout upgraded to virtio-gpu`, drives frames through
|
||||
the native backend, and **reads a pixel back** from the shared surface after a present to
|
||||
confirm the composited frame landed (`display: native present verified`), while `display-demo:
|
||||
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||
|
||||
## V5 — Mode-setting, EDID, and vsync ✅
|
||||
|
||||
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||
resource + shared surface are sized to the largest mode, so `set_mode` just re-points the
|
||||
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||
fence when the frame is on screen, which the used-ring ack the synchronous present waits on
|
||||
already gates — a tear-free present.
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasVsync` = true.
|
||||
|
||||
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||
fenced present path is exercised and confirmed (`display: vsync present ok`) — both from serial,
|
||||
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||
|
||||
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||
|
||||
- [x] The virtio-gpu driver now **hellos** the device manager (role: bus) so it is properly
|
||||
supervised — no longer stopped at the hello deadline — and is restarted on death. On
|
||||
driver loss the compositor keeps the last frame (its `.scanout` calls now return
|
||||
`-EPEER` instead of hanging — a kernel fix: an endpoint is marked dead when its owner
|
||||
dies) and **re-attaches** when the restarted driver re-announces. A permanent give-up
|
||||
(crash-loop cap) leaves the frozen frame; GOP is not re-taken.
|
||||
- [x] `test/qemu_test.py`: the `virtio-gpu`, `display-native` (hot-attach), `display-modeset`,
|
||||
and `display-reattach` (driver-kill/re-attach) cases. display-v2.md status updated.
|
||||
|
||||
**Gate (met):** the `display-reattach` case — device-manager (in `test-scanout-restart` mode)
|
||||
kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||
`supervision`, `shm`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`display-modeset`) pass; default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Client-rendered surfaces** — now unblocked by the `shm` capability (V2): an app renders
|
||||
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||
slots behind the same interface if wanted.
|
||||
- **Real-GPU (NVIDIA/AMD/Intel) drivers** — out of scope; those devices stay on the GOP
|
||||
floor by design.
|
||||
- **Hardware-accelerated compositing / multiple heads** — future.
|
||||
@@ -0,0 +1,131 @@
|
||||
# The display service v2: a pluggable scanout backend
|
||||
|
||||
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared `shm`
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced (vsync) presents, and it
|
||||
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||
|
||||
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||
composites a layer stack into a cacheable back buffer and streams damage to the linear
|
||||
framebuffer the firmware handed over. That path is portable and good: it drives any GPU,
|
||||
including a real NVIDIA card at an ultrawide's native resolution, with zero GPU-specific
|
||||
code. v2 keeps it as the **floor** and makes *scanout* — how a finished frame reaches the
|
||||
panel — a **pluggable backend**, so the compositor can **upgrade to a real GPU driver when
|
||||
one is present** and fall back to the framebuffer when it isn't.
|
||||
|
||||
The compositor itself (layers, back buffer, damage) does not change. Only the last step —
|
||||
"put this frame on screen" — becomes swappable.
|
||||
|
||||
## The shape
|
||||
|
||||
```
|
||||
compositor (display service) ── layer stack + back buffer + damage (unchanged)
|
||||
│ composites a frame, then: backend.present(damage)
|
||||
▼
|
||||
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||
│
|
||||
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||
│ Always available. No mode-set, no vsync. THE FLOOR.
|
||||
│
|
||||
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||
service: present via a shared resource + flush (real vsync),
|
||||
EDID mode list, runtime mode-set.
|
||||
```
|
||||
|
||||
A **backend** is a small interface the compositor calls:
|
||||
|
||||
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||
a virtio flush, optionally vsync-fenced, for the native path),
|
||||
- capability queries — `canModeSet`, `hasVsync` — and, when supported, `modes()` /
|
||||
`setMode(m)`.
|
||||
|
||||
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||
today; everything device-specific lives behind the interface.
|
||||
|
||||
## Selection and hot-attach
|
||||
|
||||
The choice is **dynamic**, because a GPU driver is spawned asynchronously (the device
|
||||
manager brings it up after boot), and because danos is meant to be resilient:
|
||||
|
||||
1. **Boot on GOP.** The compositor starts on `GopBackend` immediately, so there is never a
|
||||
blank screen while drivers load — the exact v1 behaviour.
|
||||
2. **Upgrade on announce.** When the virtio-gpu driver has claimed its device and set up a
|
||||
scanout, it **announces itself to the display service** (a `push`: the driver looks up
|
||||
`.display` and sends an *attach-scanout* message carrying its `scanout` endpoint as a
|
||||
capability). The compositor switches to `VirtioGpuBackend` and re-presents the current
|
||||
frame full-screen. Push beats polling — the compositor doesn't know a priori which
|
||||
driver, if any, exists, and danos has no service-registration pub/sub.
|
||||
3. **Native is restartable, not fallback-on-crash.** Once a native driver has reprogrammed
|
||||
the device, the firmware's GOP framebuffer is **stale** — "native → GOP" is not a clean
|
||||
fall-back. So a native driver that **crashes** is *restarted* by its supervisor (the
|
||||
resilience work already merged), re-announces, and the compositor **re-attaches**
|
||||
(native → native). The screen freezes on the last frame during the gap — acceptable.
|
||||
4. **GOP is the floor for "no driver was ever there."** On a real GPU (NVIDIA/AMD/Intel)
|
||||
the class-0x03 device matches nothing in the driver table, no `scanout` is ever
|
||||
announced, and the compositor stays on GOP forever — no special-casing. Only if a
|
||||
native driver *permanently* gives up (crash-loop cap) does the compositor attempt GOP
|
||||
again, and even then only if the LFB is still mappable.
|
||||
|
||||
## The shared-memory primitive this needs
|
||||
|
||||
virtio-gpu's scanout resource is **guest RAM** — the driver allocates it and attaches it
|
||||
to a virtio resource, and the compositor composes into it. That means the compositor
|
||||
writing into the driver's buffer is **cross-process memory sharing**, the primitive v1
|
||||
deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural generalization
|
||||
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||
|
||||
```
|
||||
shm_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region
|
||||
… pass `handle` as the send_cap on an ipc_call …
|
||||
shm_map(cap) -> virtual_address // the receiver maps the same physical pages
|
||||
```
|
||||
|
||||
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||
client-rendered surfaces (an app composing its own bitmap and handing the compositor a
|
||||
reference instead of drawing by command). One piece of kernel work, two features.
|
||||
|
||||
## The virtio-gpu driver
|
||||
|
||||
A new ring-3 driver process (the topology v1 anticipated — "split the driver from the
|
||||
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||
|
||||
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||
- creates a **2D scanout resource** backed by an `shm` region, `attach_backing`s it,
|
||||
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||
at a chosen mode for **runtime mode-setting**,
|
||||
- registers a `scanout` service and announces to the display service.
|
||||
|
||||
Its `resource_flush` is the real **present** — and gives a genuine **vsync/tear-free**
|
||||
path a dumb GOP framebuffer can't.
|
||||
|
||||
## What v2 unlocks — and its honest scope
|
||||
|
||||
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||
refresh / bpp), **EDID** enumeration, and **vsync**. But only on devices we have a driver
|
||||
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, vsync'd
|
||||
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||
|
||||
## Locked decisions
|
||||
|
||||
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||
present/flush (and vsync), and exercises the whole pluggable design. Tested with QEMU
|
||||
`-device virtio-gpu`.
|
||||
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||
fall-back after a reprogram.
|
||||
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
|
||||
## See also
|
||||
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||
+305
@@ -0,0 +1,305 @@
|
||||
# The display service: a framebuffer compositor
|
||||
|
||||
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||
into it directly. That console is a stop-gap. The **display service**
|
||||
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||
**presents** finished frames to the screen — the display half of the GUI track
|
||||
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||
|
||||
This note is the architecture and the reasoning behind it. The concrete build order
|
||||
lives in [display-plan.md](display-plan.md).
|
||||
|
||||
## First, a distinction that shapes everything: GOP vs. the PCI device
|
||||
|
||||
It is tempting to think "the GOP framebuffer" and "the VGA-compatible display
|
||||
controller in the PCIe tree" are two different things. They are not — they are **two
|
||||
interfaces to the same silicon, at different times and different levels**, and knowing
|
||||
which one you're holding decides what you can do.
|
||||
|
||||
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||
linear framebuffer pointer and can set video modes — but only until
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
mode list, no EDID. What survives is the frozen snapshot in
|
||||
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format}`, and nothing more.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||
triple) — but nothing binds it yet.
|
||||
|
||||
What that difference costs you, concretely:
|
||||
|
||||
| You want to… | Dumb GOP framebuffer (boot handoff) | Native device driver (PCI 0x03) |
|
||||
|-------------------------------------------|-------------------------------------|------------------------------------------|
|
||||
| **Report** the current mode | ✅ from the handoff | ✅ |
|
||||
| **Change resolution / bpp at runtime** | ❌ GOP is gone | ✅ program DISPI regs / virtio-gpu queue |
|
||||
| **Re-read EDID, enumerate monitor modes** | ❌ | ✅ the device exposes an EDID block |
|
||||
| **Refresh rate** | ❌ (virtual anyway) | only a real KMS driver — far future |
|
||||
| **vblank / tear-free present** | ❌ no vblank signal | ✅ vblank IRQ + page-flip (real GPUs) |
|
||||
| **Works on the Pi (no PCI VGA)** | ✅ VideoCore hands a simple FB | ✗ per-device |
|
||||
|
||||
The lesson: the **portable base for the whole GUI stack is the GOP / boot-handoff linear
|
||||
framebuffer**. Runtime mode-setting is a *per-device upgrade* layered on top — and on
|
||||
the Raspberry Pis there is no PCI VGA at all, so the neutral framebuffer is the only
|
||||
thing all three target machines share. That is why the display service is built on the
|
||||
dumb framebuffer first, with the native backend as an optional module behind the same
|
||||
interface.
|
||||
|
||||
## Two constraints this service exists to meet
|
||||
|
||||
Like the input service — which existed partly to motivate the asynchronous
|
||||
[`ipc_send`](ipc.md) primitive — the display service runs straight into two limits the
|
||||
rest of the system hasn't had to face:
|
||||
|
||||
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||
mapped into the kernel's physmap, and is touched only by
|
||||
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||
touch the pixels**. (See "The handoff" below — this is built.)
|
||||
|
||||
2. **danos has no cross-process shared memory.** The memory syscalls are `mmap`
|
||||
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||
use it — it would have to *map* another process's memory, which nothing allows. This
|
||||
is deferred (see "What v1 does not do"), because v1 sidesteps it entirely.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
kernel ── owns the boot framebuffer; bootstrap console only
|
||||
│ seeds a "display0" device node from BootInformation.framebuffer
|
||||
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||
│ plus DisplayInfo{width, height, pitch, format})
|
||||
▼
|
||||
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||
│ device.claim(display0) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE list
|
||||
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||
│ backend is an INTERNAL interface: {gop-fb} today; {bochs-dispi, virtio-gpu} later
|
||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
runtime.display commands: runtime.display surfaces:
|
||||
create_layer / configure_layer shm_create → pass as a capability →
|
||||
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||
present client-rendered bitmap directly
|
||||
```
|
||||
|
||||
The bring-up sequence mirrors a hardware driver's — it is the
|
||||
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||
`extern struct` messages and an `Operation` tag).
|
||||
|
||||
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||
composites — it does not split a "framebuffer driver" from a "compositor" the way input
|
||||
splits `ps2-bus` from the input service. The backend (dumb FB vs. a native GPU) is an
|
||||
*internal* interface, not a process boundary. That boundary earns its keep only when a
|
||||
second backend or a second monitor appears; until then it is complexity with no payoff.
|
||||
|
||||
## The handoff: a device node + a write-combining map
|
||||
|
||||
The framebuffer crosses into user space through the machinery that already exists for
|
||||
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||
re-claims it).
|
||||
|
||||
- The kernel seeds a synthetic **`display0`** node into the
|
||||
[devices-broker](../system/kernel/devices-broker.zig) at init, from
|
||||
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||
`DisplayInfo{width, height, pitch, format}` (the memory resource says *where* and *how
|
||||
big*; `DisplayInfo` says how to *interpret* the bytes).
|
||||
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||
back→front blit unusably slow.
|
||||
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||
LFB. A panic is the one exception — by then the service is likely dead anyway, and a
|
||||
panic on screen wins.
|
||||
|
||||
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||
`vfs`/`input`/`device-manager` ([init.zig](../system/services/init/init.zig)), and it
|
||||
self-discovers `display0` with `device.enumerate`. The [device manager](device-manager.md)
|
||||
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||
this singleton synthetic node.
|
||||
|
||||
## Double buffering and the write-combining discipline
|
||||
|
||||
Two buffers, with deliberately different memory types:
|
||||
|
||||
- The **front buffer** is the LFB — **write-combining**: fast to *write*, slow to
|
||||
*read*. The rule is therefore **never read the front buffer**. Only ever stream into
|
||||
it, sequentially.
|
||||
- The **back buffer** is ordinary **cacheable** RAM (`mmap`), the same geometry. All
|
||||
compositing happens here, where reads and read-modify-write blends are cheap.
|
||||
|
||||
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||
|
||||
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||
|
||||
These are two different artifacts, and the dumb framebuffer fixes exactly one of them:
|
||||
|
||||
- **Flicker** is the user seeing intermediate, half-drawn states (a clear-then-redraw
|
||||
flash). Double buffering **eliminates it completely** — the screen only ever receives
|
||||
whole, finished frames.
|
||||
- **Tearing** is a present landing while the display's scanout beam is mid-frame, so the
|
||||
top of the screen shows the new frame and the bottom the old. Avoiding it requires
|
||||
presenting during the vertical blank (**vsync**) — which needs a vblank signal. **A
|
||||
dumb GOP framebuffer has no vblank.**
|
||||
|
||||
So v1 is **flicker-free**, and it *minimizes* the tear window by presenting only damaged
|
||||
rectangles (less to copy → a smaller window in which the beam can catch a half-updated
|
||||
frame), but it is **not tear-free**. Genuine vsync waits for a backend with a vblank IRQ
|
||||
or a flush/flip path — a native-device capability, not something the firmware
|
||||
framebuffer can offer. Stated plainly here so the limitation is understood, not
|
||||
discovered.
|
||||
|
||||
## Layers and the client protocol
|
||||
|
||||
The compositor holds an **ordered stack of layers**. Each layer has a rectangle, a
|
||||
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||
|
||||
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||
immediate-mode command protocol — essentially the model early X used, and enough for a
|
||||
shell, a terminal, a cursor, and a wallpaper:
|
||||
|
||||
| Operation | Meaning |
|
||||
|--------------------|---------------------------------------------------------------|
|
||||
| `info` | report `{width, height, pitch, format}` of the display |
|
||||
| `create_layer` | allocate a server-owned surface, return a layer handle |
|
||||
| `configure_layer` | set a layer's rect, z-order, visibility |
|
||||
| `destroy_layer` | release a layer |
|
||||
| `fill_rect` | fill a rectangle of a layer with a colour |
|
||||
| `blit_tile` | copy a small client-supplied pixel tile into a layer (inline) |
|
||||
| `damage` | mark a region of a layer dirty |
|
||||
| `present` | composite dirty layers and flush to the screen |
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
policy in the client.
|
||||
|
||||
## `runtime.display`
|
||||
|
||||
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||
runtime, as with every other danos service.
|
||||
|
||||
## The cursor: a mouse-listener thread feeding the compositor
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
- **Listener thread.** Blocks on the input service's mouse stream
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`runtime.Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||
message-notification ([ipc.md](ipc.md)). The poke is *coalesced*: at most one is queued
|
||||
while the main loop has not drained the last, so a fast mouse cannot flood the endpoint.
|
||||
- **Render.** On the poke, the main loop takes the latest position and moves the cursor —
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
||||
both are clean additions behind the interfaces v1 establishes.
|
||||
|
||||
- **Client-rendered surfaces (shared memory).** The fast path for a bitmap-heavy app is
|
||||
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||
from *endpoints* to *memory objects* (`shm_create(len) → {cap, virtual_address}`, pass `cap` on
|
||||
an `ipc_call`, receiver `shm_map(cap) → virtual_address`). v1 avoids it because server-owned
|
||||
surfaces already prove the whole pipeline.
|
||||
|
||||
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||
resolution / bpp at runtime needs the raw PCI device. The first native backend is
|
||||
Bochs DISPI — the register interface QEMU's `-device VGA` exposes — behind the same
|
||||
internal backend interface the dumb framebuffer sits behind. Refresh-rate and colour
|
||||
management (a gamma LUT) are real-GPU-KMS territory, far beyond this.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
test/qemu_test.py <case>`), each layering on the last:
|
||||
|
||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||
the claim → `mmio_map` leaf is genuinely **write-combining** (PAT entry 4), asserted at
|
||||
the page-table level.
|
||||
- **`display-service`** — the compositor comes up: it claims the framebuffer, allocates
|
||||
the cacheable back buffer, presents a cleared frame through the double-buffer path
|
||||
(`display: online … / presented frame 0`), and a startup **self-check** composites two
|
||||
overlapping layers on the real framebuffer and reads them back — overlap = the top
|
||||
layer — logging `display: compositor self-check ok`.
|
||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||
[`display-demo`](../system/services/display-demo/) client (the
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
a sliding rectangle — through the layer client API and heartbeats
|
||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||
no cursor and reads no input — the cursor is the service's own (below), and the demo
|
||||
animates on its own frame timer, independent of the mouse (the test spawns `input`
|
||||
alongside it to keep that independence honest). The visible motion itself is a screenshot
|
||||
away via `zig build run-x86-64`.
|
||||
- **`display-cursor`** — the mouse-listener thread end to end: with the `input` service up,
|
||||
`input-source mouse` publishes pure motion, and the display's listener thread accumulates
|
||||
it into a cursor position handed to the render loop over the `CursorChannel`. Once the
|
||||
cursor has tracked a run of that motion, the service logs
|
||||
`display: cursor tracking mouse ok`. Runs `smp: 4` — the compositor and listener threads
|
||||
execute on different cores, which is what surfaced the IPC-under-lock requirement above.
|
||||
|
||||
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
## See also
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||
- [display-plan.md](display-plan.md) — the ordered build-out.
|
||||
@@ -0,0 +1,368 @@
|
||||
# The driver model: buses, classes, and host controllers
|
||||
|
||||
[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its
|
||||
registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is
|
||||
not enough for a disk, a keyboard, or a network card, because those hang off a
|
||||
*controller*, on a *bus*, speaking a *protocol*, and no single process should have to
|
||||
know all three.
|
||||
|
||||
Real driver stacks factor into three shapes. This document is about what each one is,
|
||||
what the kernel must give it, how they share code — and precisely which primitive each
|
||||
is still blocked on.
|
||||
|
||||
## Three shapes
|
||||
|
||||
| Shape | Owns | Reaches hardware by | Talks to |
|
||||
|---|---|---|---|
|
||||
| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language |
|
||||
| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC |
|
||||
| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC |
|
||||
|
||||
The last row is the surprising one and the whole point. A USB keyboard driver touches
|
||||
no registers, takes no interrupts, and maps no memory. It sends HID protocol messages
|
||||
to whatever published the device, and it works identically whether the controller
|
||||
below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that
|
||||
outlive the hardware they were written for.
|
||||
|
||||
In practice **HCD and bus driver are usually the same process**. An xHCI driver is a
|
||||
host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA
|
||||
rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting
|
||||
them is a fiction; what matters is that both *roles* have kernel support, because a
|
||||
plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
|
||||
## The device table is the spine
|
||||
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||
`mmio_map`, `irq_bind`, `device_register` — checks `devices_broker.ownerOf(id) == me`.
|
||||
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||
`device_register` cannot be a free-for-all.
|
||||
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||
resource of the same kind on its parent (`devices_broker.contains`). A bus driver can only
|
||||
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||
syscall named "map any physical page you like."
|
||||
|
||||
Containment is transitive by construction: a grandchild is contained in its child,
|
||||
which is contained in the bus. Nothing can be laundered through a chain.
|
||||
|
||||
Note that firmware topology does **not** obey containment, and isn't asked to — a PCI
|
||||
function's BAR is not inside its host bridge's `bus_range`, because a bus-number range
|
||||
is not an address window. Discovery is trusted; user space is not.
|
||||
|
||||
### What a bus driver looks like
|
||||
|
||||
danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and
|
||||
`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block
|
||||
as the "bus" and its comparators as the "devices", is:
|
||||
|
||||
```zig
|
||||
_ = dev.claim(bus.id); // 1. own the bus
|
||||
const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware
|
||||
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||
|
||||
for (0..n) |i| { // 3. publish each child
|
||||
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = memory,
|
||||
.start = bus_mmio.start + 0x100 + 0x20 * i,
|
||||
.len = 0x20 };
|
||||
_ = dev.register(bus.id, &child).?; // kernel checks containment
|
||||
}
|
||||
```
|
||||
|
||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||
whose window escapes the bus is refused; the in-kernel `containment` test asserts the
|
||||
kernel's table upholds that ([drivers.md](drivers.md)).
|
||||
|
||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||
|
||||
## Families: sharing code between drivers
|
||||
|
||||
A "family" is two modules, not one:
|
||||
|
||||
- **A logic module** — the parts of the bus that every driver on it re-derives. Config
|
||||
space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub
|
||||
protocol for USB.
|
||||
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||
*whatever* published its device. This is the part that makes class drivers portable.
|
||||
|
||||
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||
[`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
|
||||
and its clients. The pattern generalises directly:
|
||||
|
||||
```
|
||||
library/
|
||||
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||
bus/
|
||||
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||
usb/ module "usb" — descriptors, control transfers, hubs
|
||||
proto/
|
||||
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||
block/ module "block-protocol"
|
||||
hid/ module "hid-protocol"
|
||||
|
||||
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||
block/ class driver imports runtime, block-protocol
|
||||
```
|
||||
|
||||
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||
everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
|
||||
## What exists today
|
||||
|
||||
- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device
|
||||
grants, `device_grant` teardown.
|
||||
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||
EOI; `irq_ack` is the unmask.
|
||||
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||
- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument
|
||||
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||
no DMA driver consumes `dma_alloc` yet.
|
||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||
with no new syscall), and `msi_bind(device_id, endpoint) -> address, data` allocates a
|
||||
per-device edge-triggered vector, delivered as an IPC notification with no mask and no
|
||||
ack cycle. Legacy INTx (`_PRT` parsing + shared lines) is deliberately skipped — MSI
|
||||
is the real answer. QEMU's HPET has no MSI, so delivery is proven with a self-IPI; the
|
||||
first PCI driver is the first real consumer.
|
||||
- **Port I/O** — `io_read`/`io_write(device_id, resource_index, offset, width[, value])`:
|
||||
a claimed device's `io_port` resource lets a driver read/write its ports, gated exactly
|
||||
like `mmio_map` gates memory (direct ring-3 `in`/`out` stays a #GP). This is what makes
|
||||
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||
they're used.
|
||||
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||
platform info). This is detection only — **no translation domains are programmed, so
|
||||
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||
booted with an emulated `intel-iommu`.
|
||||
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||
([sysv.md](sysv.md)). This is what
|
||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||
spawn capability is future work.
|
||||
|
||||
So: **bus drivers work now, and they're started by the device manager, not the kernel.**
|
||||
HCDs and class drivers do not work yet. Here is exactly why, and exactly what would fix it.
|
||||
|
||||
---
|
||||
|
||||
# Proposed ABI
|
||||
|
||||
## M13 — capability passing, for class drivers ✅ done
|
||||
|
||||
*Implemented as described below (see "What exists today"). The signatures landed
|
||||
verbatim: `send_cap` in r9, `received_cap` returned in r8, `-ENOSPC` on a full receiver
|
||||
table with no delivery. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker.** A class driver has to reach *its* device. Today the only way to find
|
||||
an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`,
|
||||
where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot
|
||||
mint one endpoint per USB device that way, and there is no way for a bus driver to
|
||||
*hand* a class driver an endpoint. M7 deferred this deliberately.
|
||||
|
||||
**The fix.** Let a message carry one handle. Sender names a handle in its own table;
|
||||
the kernel installs the endpoint into the receiver's table (bumping `refcount`) and
|
||||
tells the receiver the index it landed at.
|
||||
|
||||
```
|
||||
ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len
|
||||
ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap)
|
||||
-> recv_len (rax), badge (rdx), received_cap (r8)
|
||||
```
|
||||
|
||||
`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred
|
||||
endpoint was installed at in the receiver's table, or `no_cap`.
|
||||
|
||||
- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`,
|
||||
leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`;
|
||||
this needs a third (`setSyscallResult3`).
|
||||
- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is
|
||||
not delivered** — a half-delivered capability is worse than a failed send.
|
||||
- `closeHandles` already drops references on exit, so the lifetime story is unchanged.
|
||||
|
||||
That single primitive gives you the standard `open` pattern:
|
||||
|
||||
```zig
|
||||
// class driver // bus driver
|
||||
const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...);
|
||||
const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||
.{ .op = .open, .id = dev_id }); // reply with it as send_cap
|
||||
// now dev_ep is a private channel to that one device
|
||||
```
|
||||
|
||||
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||
|
||||
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
|
||||
device can read, which means memory that is (a) physically contiguous, (b) at a
|
||||
physical address the driver knows, (c) of the right cacheability, and (d) pinned.
|
||||
[`sysMmap`](system/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
|
||||
once per page, maps writeback-cached, and never reveals a physical address.
|
||||
|
||||
**The fix.**
|
||||
|
||||
```
|
||||
dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx)
|
||||
dma_free(virtual_address, len) -> 0
|
||||
|
||||
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||
dma_below_4g (4) for devices with 32-bit DMA addressing
|
||||
```
|
||||
|
||||
Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the
|
||||
mapping, and the physical address is stable. It needs one thing the kernel lacks —
|
||||
`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time
|
||||
with no adjacency guarantee.
|
||||
|
||||
**The memory-ordering contract.** danos has, at the time of writing, **zero memory
|
||||
barriers anywhere in the tree.** That is currently correct-by-accident and won't
|
||||
survive the first DMA driver, or the first ARM boot.
|
||||
|
||||
`volatile` is not a barrier. In Zig it means: don't elide this access, and don't
|
||||
reorder it against *other volatile* accesses. It says nothing about your *ordinary*
|
||||
stores — the descriptor you just filled in normal WB memory — which LLVM may freely
|
||||
sink past a volatile MMIO write. The canonical bug:
|
||||
|
||||
```zig
|
||||
ring[i] = descriptor; // ordinary store to WB RAM
|
||||
doorbell.* = i; // volatile store to UC MMIO
|
||||
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||
```
|
||||
|
||||
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||
|
||||
| Situation | Required |
|
||||
|---|---|
|
||||
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||
| MMIO write that must complete before the next read | `mb()` |
|
||||
|
||||
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||
sprinkling of `asm volatile`:
|
||||
|
||||
| | x86_64 | aarch64 |
|
||||
|---|---|---|
|
||||
| `mb()` | `mfence` | `dsb sy` |
|
||||
| `rmb()` | `lfence` | `dsb ld` |
|
||||
| `wmb()` | `sfence` | `dsb st` |
|
||||
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||
|
||||
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||
condition. Build the abstraction while there is one caller to fix.
|
||||
|
||||
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||
|
||||
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||
|
||||
*Implemented the MSI half: ECAM config space per PCI function (resource 0) and
|
||||
`msi_bind` (per-device edge-triggered vector, delivered as a notification). Legacy INTx
|
||||
`_PRT` parsing is skipped on purpose. `msi_bind` returns (address, data) as two values
|
||||
rather than an out-struct. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||
device has.
|
||||
|
||||
**The fix, in two halves.**
|
||||
|
||||
*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it
|
||||
as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**,
|
||||
and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the
|
||||
line must be polled on each interrupt — the reason everyone left INTx behind.
|
||||
|
||||
*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no
|
||||
mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver
|
||||
the (address, data) pair to program into its own MSI capability:
|
||||
|
||||
```
|
||||
msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 }
|
||||
```
|
||||
|
||||
The driver writes those into config space itself — which means it needs config space,
|
||||
which means **discovery should give each `pci_device` a `.memory` resource for its
|
||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||
|
||||
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||
building the per-device domains alongside that first driver is both the natural order
|
||||
and the only way to verify them. The rest of this section is the original caveat.*
|
||||
|
||||
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||
A driver that can program a bus-mastering engine can make that device write to any
|
||||
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||
on any DMA-capable device is equivalent to granting ring 0.**
|
||||
|
||||
This does not make the model useless — it's the same position Linux is in with the
|
||||
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||
gap should be named rather than implied.
|
||||
|
||||
## Ordering
|
||||
|
||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||
could write a real one against any device a bus driver publishes tomorrow.
|
||||
|
||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||
from hand-rolling `*volatile` and getting ARM wrong.
|
||||
|
||||
## See also
|
||||
|
||||
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
+395
@@ -0,0 +1,395 @@
|
||||
# Writing a driver
|
||||
|
||||
In a monolithic kernel a driver is a function call away from everything: it runs in
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](resilience.md)).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||
([discovery](discovery.md), [acpi](acpi.md)).
|
||||
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||
device's physical MMIO window into your address space, and from then on it's plain
|
||||
memory. No syscall per register access.
|
||||
3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered
|
||||
to you as an IPC notification. You block; the hardware wakes you.
|
||||
|
||||
A driver is, in one sentence, *a process that sleeps until its device has something to
|
||||
say.*
|
||||
|
||||
## How a driver gets started: discover, match, spawn
|
||||
|
||||
Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that
|
||||
is policy, and policy lives in user space. Boot brings user space up as a three-level
|
||||
supervision hierarchy, each level owning one job:
|
||||
|
||||
```
|
||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus
|
||||
| | |
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
initial-ramdisk services (vfs, the to a driver, and system_spawn's it
|
||||
so user space can device-manager). Its
|
||||
system_spawn from it list is init policy.
|
||||
```
|
||||
|
||||
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
||||
ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled
|
||||
in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0],
|
||||
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||
|
||||
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||
the `device-manager` — from a small list. Drivers are deliberately *not* its job.
|
||||
- **device-manager** ([system/services/device-manager](system/services/device-manager/device-manager.zig))
|
||||
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||
its probe path, entirely from ring 3:
|
||||
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||
ACPI/PCI ([discovery](discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver by `DeviceClass`. The match policy
|
||||
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||
describing its own match).
|
||||
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||
arguments can carry *which* device it matched), which then claims
|
||||
its device and runs the event loop below.
|
||||
|
||||
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||
capability yet.
|
||||
|
||||
## The capability: claim before touch
|
||||
|
||||
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
|
||||
| # | Call | Meaning |
|
||||
|---|------|---------|
|
||||
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||
| 13 | `mmio_map(id, res_idx) -> virtual_address` | Map a claimed device's register window |
|
||||
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||
|
||||
Notice that **nothing takes a physical address or an interrupt number.** Every call
|
||||
names a device by id and a resource by index. That indirection is the entire security
|
||||
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||
the same check at the top of `sysMmioMap`):
|
||||
|
||||
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||
for `irq_bind`
|
||||
|
||||
The claim is the capability. Everything else follows from it.
|
||||
|
||||
## Registers: `mmio_map`
|
||||
|
||||
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||
with `present | user | writable | nx | pcd | pwt`
|
||||
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||
|
||||
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||
would return a stale value and a write might never leave the CPU.
|
||||
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||
the frame allocator as if they were free RAM. The `iopass` test guards it.
|
||||
|
||||
Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages
|
||||
never widen an existing mapping.
|
||||
|
||||
Then you just… use it:
|
||||
|
||||
```zig
|
||||
const base = dev.mmioMap(dev_id, mmio_res) orelse return;
|
||||
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
||||
const now = counter.*; // a load, straight to the hardware. no kernel involved.
|
||||
```
|
||||
|
||||
## Interrupts: the cycle, and why it has that shape
|
||||
|
||||
An interrupt handler in a microkernel has a problem. The code that knows how to quiet
|
||||
the device is in ring 3, in another address space, and it will not run for
|
||||
microseconds or milliseconds — after a context switch, when the scheduler gets to it.
|
||||
But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until
|
||||
the device is quieted. EOI a still-asserted line and the I/O APIC redelivers
|
||||
immediately. Forever. The driver never gets to run at all.
|
||||
|
||||
The way out is to mask the line before acknowledging it:
|
||||
|
||||
```
|
||||
kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU
|
||||
irqEoi() // now safe to tell the LAPIC we're done
|
||||
notifyFromIsr() // wake the driver — it runs much later
|
||||
|
||||
driver replyWait() -> badge with the notify bit set
|
||||
<clear the device's status register> // NOW the line deasserts
|
||||
irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire
|
||||
|
||||
```
|
||||
|
||||
`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the
|
||||
interrupt fires exactly once, ever; call it before the device is quiet and you get an
|
||||
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||
syscalls and not one.
|
||||
|
||||
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||
handler knows which discipline its source needs.
|
||||
|
||||
### The driver side is an event loop, not a callback
|
||||
|
||||
`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by
|
||||
the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one
|
||||
single-threaded loop over both of its event sources:
|
||||
|
||||
```zig
|
||||
while (true) {
|
||||
const r = ipc.replyWait(endpoint, reply, &recv);
|
||||
if (r.isNotification()) { // r.source() is the GSI
|
||||
service_device(); // clear the status register
|
||||
_ = dev.irqAck(id, irq_res); // re-arm
|
||||
} else {
|
||||
handle_client_request(recv[0..r.len]);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
No reentrancy, no "what am I allowed to call from an interrupt handler", no shared
|
||||
state between ISR and task context. The interrupt is just a message.
|
||||
|
||||
Two properties worth knowing:
|
||||
|
||||
- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in
|
||||
an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody
|
||||
waiting, but the badge is already on the endpoint's notify ring. The next
|
||||
`replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO).
|
||||
- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on
|
||||
overflow. That's correct: an IRQ notification is a *level* ("the device wants
|
||||
attention"), not a tally. Re-read the device's status register; never assume one
|
||||
notification means exactly one event. Because the ISR masks the line until you
|
||||
`irq_ack`, at most one badge per GSI can be outstanding — so the ring can only
|
||||
overflow if you bind more than eight GSIs to a single endpoint. Don't.
|
||||
|
||||
## A whole driver
|
||||
|
||||
A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such
|
||||
example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`,
|
||||
`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a
|
||||
compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the
|
||||
shape is:
|
||||
|
||||
```zig
|
||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||
// class=timer with memory + irq
|
||||
_ = dev.claim(hpet.dev_id); // the capability
|
||||
const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?;
|
||||
const endpoint = ipc.createIpcEndpoint().?;
|
||||
|
||||
// program the hardware over the mapping we were just handed
|
||||
reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config
|
||||
reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator
|
||||
reg(base, 0x010).* |= 1; // ENABLE
|
||||
|
||||
_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint);
|
||||
|
||||
while (...) {
|
||||
const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling.
|
||||
if (r.badge & notify_bit == 0) continue;
|
||||
reg(base, 0x020).* = 1; // clear status -> deassert
|
||||
reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm
|
||||
_ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask
|
||||
}
|
||||
```
|
||||
|
||||
The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a
|
||||
clocksource — the only way to use it is to read it, so it exercises `mmio_map` without
|
||||
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||
mask/ack cycle above is exercised for real rather than being decoration on an
|
||||
edge-triggered line that would have been fine without it.
|
||||
|
||||
One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**.
|
||||
Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in
|
||||
the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the
|
||||
mask, and records one concrete GSI as an `irq` resource. The driver then programs
|
||||
`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one
|
||||
it recorded. Hardware that describes itself at runtime still has to fit through a
|
||||
static capability.
|
||||
|
||||
## Publishing children: `device_register`
|
||||
|
||||
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||
That's `device_register`, and it makes the device table a tree rather than a list
|
||||
(`DeviceDesc.parent`).
|
||||
|
||||
```zig
|
||||
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||
const child_id = dev.register(bus_id, &child).?;
|
||||
```
|
||||
|
||||
The child is left **unclaimed**, which is the whole point: another process claims it and
|
||||
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||
|
||||
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||
bus driver may only ever subdivide what it already owns.
|
||||
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||
drivers and host controller drivers fit together.
|
||||
|
||||
## What the kernel does not do for you
|
||||
|
||||
- **It does not quiet your device.** That's the whole reason `irq_ack` exists.
|
||||
- **It does not know your registers.** `mmio_map` hands you a base address; every
|
||||
offset in this document came from the HPET spec, not from danos.
|
||||
- **It does not serialise your driver.** Two clients calling one driver endpoint are
|
||||
serialised by `replyWait`, but nothing stops your driver from being preempted.
|
||||
|
||||
## Limits, today
|
||||
|
||||
Worth knowing before you write the second driver:
|
||||
|
||||
Several things this list used to warn about are now available (see
|
||||
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||
uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
`mb`/`rmb`/`wmb`). What remains:
|
||||
|
||||
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||
than a page, but its *mapping* can't.
|
||||
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||
that device write to *any* physical address — page tables don't sit between a device
|
||||
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||
what it delivers; enforcement lands with the first DMA driver.
|
||||
- **No `dev_release`.** A claim is never dropped (only IRQ/MSI bindings are, on exit), so
|
||||
a device stays owned for the life of its driver — which blocks restart.
|
||||
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||
- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override
|
||||
says active-low needs that threaded through from discovery.
|
||||
- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and
|
||||
by a single I/O APIC.
|
||||
- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops
|
||||
on overflow. With one GSI per endpoint that's unreachable — the line is masked from
|
||||
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||
left to ack it.
|
||||
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||
[not built yet](resilience.md).
|
||||
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||
line on exit, but the claim is never released — restart is
|
||||
[not built](resilience.md).
|
||||
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
|
||||
back. See the note at the top of `system/kernel/irq.zig`.
|
||||
|
||||
## Verifying it
|
||||
|
||||
No demo driver ships to prove this end to end; the *real* drivers do, so the tests
|
||||
target them and the kernel primitives directly:
|
||||
|
||||
- **`device-manager`** — boots only the device manager, which discovers the PCI host
|
||||
bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the
|
||||
process table and the device tree — to confirm pci-bus came up and registered the
|
||||
functions it enumerated: the whole discover → match → spawn → driver-up chain.
|
||||
- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ,
|
||||
delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end.
|
||||
- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM
|
||||
window) and walks it: `mmio_map`, end to end.
|
||||
- **`containment`** — the kernel refuses a `device_register` whose child window escapes
|
||||
the parent's grant (else it would be a syscall for mapping arbitrary memory), while an
|
||||
identical re-register stays idempotent. Asserted in-kernel, straight against the broker.
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
is not. That second half is why bindings are keyed on the owning *task* and not on the
|
||||
endpoint pointer — endpoints are shared, so releasing "everything pointing at this
|
||||
endpoint" would silently mask a live driver's device.
|
||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||
space never returns MMIO frames to the RAM pool.
|
||||
|
||||
```
|
||||
$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass
|
||||
device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
acpi-ps2 ... PASS
|
||||
pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
containment ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
```
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||
the first DMA driver to protect and test against) and these smaller items:
|
||||
|
||||
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||
which blocks restart.
|
||||
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||
table.
|
||||
- **Restart.** A supervisor that *spawns* drivers now exists — the device-manager starts
|
||||
them with `system_spawn` — but a supervisor that *restarts* them does not. A driver that
|
||||
dies should release its claim, have its device quiesced, and be respawned; today nothing
|
||||
notices the death. Some pieces (`releaseIrqs`, `device_grant` teardown, the claim table)
|
||||
exist, and `dev_release` (below) is the missing mechanism; the restart policy is the
|
||||
resilience track ([resilience.md](resilience.md)).
|
||||
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||
a woken driver waits for the next scheduling point.
|
||||
|
||||
## The driver contract (M17–M18)
|
||||
|
||||
Claiming and mapping is half of being a danos driver; the other half is the
|
||||
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||
|
||||
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
([device-manager.md](device-manager.md); usb-xhci-bus is the reference
|
||||
implementation).
|
||||
- Crash freely — that is the design. The kernel releases your claims, IRQ
|
||||
bindings, and MSI vectors at death; the manager reads your exit reason,
|
||||
prunes what you reported, restarts you with backoff, and your fresh instance
|
||||
re-claims and re-reports. Never depend on your own cleanup running
|
||||
(iron rule 1).
|
||||
+25
-20
@@ -10,7 +10,7 @@ that hands us a working CPU, a memory map, and a screen, and then gets out of th
|
||||
way.
|
||||
|
||||
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
||||
exists to load *us*. Our `src/boot/efi.zig` is a UEFI *application* — a normal program
|
||||
exists to load *us*. Our `boot/efi.zig` is a UEFI *application* — a normal program
|
||||
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
||||
and then jump into the kernel.
|
||||
|
||||
@@ -20,14 +20,17 @@ UEFI boots by looking for a FAT-formatted partition called the **EFI System
|
||||
Partition (ESP)** and running a file at a well-known fallback path:
|
||||
|
||||
```
|
||||
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the
|
||||
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
|
||||
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
|
||||
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
|
||||
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
|
||||
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||
|
||||
## Boot services: the firmware's API
|
||||
|
||||
@@ -83,9 +86,9 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
||||
|
||||
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
||||
**SimpleFileSystem** to open that volume.
|
||||
- Open the file named `danos`, seek to the end to learn its size, rewind, and read
|
||||
the whole ELF into a firmware-allocated pool buffer. (`read` may return short, so
|
||||
we loop.)
|
||||
- Open the kernel ELF at its FHS path (`system\kernel`), seek to the end to learn its
|
||||
size, rewind, and read the whole ELF into a firmware-allocated pool buffer. (`read`
|
||||
may return short, so we loop.)
|
||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||
walk the program headers. For every `PT_LOAD` segment we:
|
||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||
@@ -121,7 +124,7 @@ entirely ours.
|
||||
### 4. Jump to the kernel
|
||||
|
||||
```zig
|
||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn =
|
||||
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||
@ptrFromInt(entry);
|
||||
kernel(&boot_info);
|
||||
```
|
||||
@@ -139,17 +142,19 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
||||
read garbage.
|
||||
|
||||
So both sides pin the convention explicitly to SysV via the shared
|
||||
`danos.kernel_abi` (defined in `src/root.zig`). The loader's function-pointer type
|
||||
and the kernel's `_start` both reference it, so the pointer lands in the register
|
||||
the kernel expects. This is the whole reason `kernel_abi` lives in the shared
|
||||
`danos` module: it's a contract both binaries must agree on. See
|
||||
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||
|
||||
## The handoff contract
|
||||
|
||||
The loader and kernel are two *separate* binaries built for two different targets,
|
||||
so everything they exchange must have an identically-defined memory layout. That's
|
||||
what `src/root.zig` provides — imported by both as the `danos` module:
|
||||
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
|
||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||
framebuffer; this is where future handoff data like the memory map will go).
|
||||
@@ -164,16 +169,16 @@ the loader writes are the bytes the kernel reads.
|
||||
```
|
||||
power on
|
||||
-> UEFI firmware initialises hardware
|
||||
-> finds esp/EFI/BOOT/BOOTX64.efi, runs it (our efi.zig main)
|
||||
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||
-> grab boot services
|
||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||
-> exitBootServices (retry until the memory-map key holds)
|
||||
-> jump to e_entry, boot_info pointer in RDI
|
||||
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt)
|
||||
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||
```
|
||||
|
||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||
disappear.** `src/boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||
disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
||||
on our own.
|
||||
|
||||
@@ -3,12 +3,12 @@
|
||||
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
||||
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
||||
here's one back. That's the **physical frame allocator** (a "physical memory
|
||||
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||
manager", hence `system/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||
natural unit because that's the granularity the CPU's paging hardware maps — and
|
||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||
per-process memory all ultimately ask the frame allocator for pages.
|
||||
|
||||
It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion`
|
||||
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||
|
||||
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
|
||||
|
||||
## How it works
|
||||
|
||||
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||
State lives in `system/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||
and a `next_hint` marking where the next allocation scan should start.
|
||||
|
||||
### init(map) — building it from the memory map
|
||||
|
||||
+2
-2
@@ -9,10 +9,10 @@ write a 32-bit value to the right address, and a pixel changes color. That's
|
||||
exactly what `Console.pixel` does:
|
||||
|
||||
```zig
|
||||
self.rowPtr(y)[x] = color; // src/kernel/console.zig
|
||||
self.rowPtr(y)[x] = color; // system/kernel/console.zig
|
||||
```
|
||||
|
||||
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to
|
||||
Our `Framebuffer` struct (`system/boot-handoff.zig`) is the four facts you need to
|
||||
address it:
|
||||
|
||||
| Field | Meaning |
|
||||
|
||||
+3
-3
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
|
||||
## The core of it: `hlt`
|
||||
|
||||
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
||||
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||
generic kernel calls it as `arch.halt()`:
|
||||
|
||||
```zig
|
||||
@@ -83,7 +83,7 @@ treats the call:
|
||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||
jumps back out.
|
||||
|
||||
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls
|
||||
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||
`noreturn`. The "never returns" property is threaded all the way down.
|
||||
|
||||
@@ -106,7 +106,7 @@ There are three halt sites, and they're all the same idea:
|
||||
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
||||
than limping on with corrupted state — is the safe response.
|
||||
|
||||
3. **Bootloader failure** — in `src/boot/efi.zig`, if `boot()` fails *before* handing
|
||||
3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
|
||||
off to the kernel, `main` logs the error and parks the machine with the same
|
||||
loop so the message stays on screen:
|
||||
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
|
||||
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
||||
eventually a process table.
|
||||
|
||||
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is
|
||||
It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
|
||||
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
||||
|
||||
## A growable free-list allocator
|
||||
|
||||
+161
@@ -0,0 +1,161 @@
|
||||
# The input module: broadcasting input events
|
||||
|
||||
A keyboard driver has one keystroke and *many* programs that might want it — a shell, a
|
||||
window server, a logger. None of them owns the hardware, and the driver should not know
|
||||
who is listening. So between the drivers and the listeners sits the **input service**
|
||||
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||
reached over IPC, like the [VFS server](../system/services/vfs/vfs.zig) — no kernel knows
|
||||
what a key is.
|
||||
|
||||
## One service, several device classes
|
||||
|
||||
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||
**joystick/gamepad** — and is built to take more
|
||||
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||
event:
|
||||
|
||||
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||
produced, carrying the Unicode scalar); plus a layout-independent `keycode` and a
|
||||
`modifiers` bitmask.
|
||||
- `MouseEvent` — relative `motion` (`dx`/`dy`), `button_down`/`button_up`, and `scroll`.
|
||||
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||
`button_down`/`button_up`.
|
||||
|
||||
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||
mouse-only listener never wakes for keystrokes.
|
||||
|
||||
## Why this needed a new kernel primitive
|
||||
|
||||
The interesting part is delivery, and it runs straight into the shape of danos IPC.
|
||||
[ipc.md](ipc.md) describes a **synchronous rendezvous**: a server holds exactly one
|
||||
pending reply (`Task.ipc_client`) and *must* answer it on its next `replyWait`. Two
|
||||
consequences decide the whole design:
|
||||
|
||||
1. **You cannot block N subscribers waiting for "the next event".** A server can hold only
|
||||
one caller at a time, so the natural "subscriber calls `next_event()` and blocks" API
|
||||
is impossible for more than one subscriber. Delivery therefore has to be **push** — the
|
||||
service reaching out to subscribers — not pull.
|
||||
|
||||
2. **A synchronous push can hang the whole service.** If the service delivered with
|
||||
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||
the kernel does **not** wake a caller parked on a *dead* peer's endpoint (it only fails a
|
||||
peer that was mid-reply — see [process.zig](../system/kernel/process.zig)
|
||||
`releaseTaskResourcesLocked`). One subscriber that exits mid-delivery would wedge input
|
||||
for everyone. That is the opposite of the resilience the microkernel is for.
|
||||
|
||||
The fix is the asynchronous send that [ipc.md](ipc.md) had already earmarked as future
|
||||
work ("asynchronous / buffered send … for notifications between servers"):
|
||||
|
||||
```
|
||||
ipc_send(handle, message_ptr, message_len) -> 0 / -errno
|
||||
```
|
||||
|
||||
`ipc_send` copies a small payload into the endpoint's **bounded queue** and wakes a
|
||||
receiver, then returns immediately — it never blocks and so can never hang on a dead or
|
||||
slow subscriber. The receiver picks it up through the same `replyWait` it already runs:
|
||||
the wake arrives as a **buffered message** — `notify_badge_bit | notify_message_bit` set in
|
||||
the badge (distinguishing it from a bare IRQ/child-exit notification), the sender's task id
|
||||
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||
discrete data, not a coalescing "level" like an interrupt. See
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
the `replyWait` receive loop).
|
||||
|
||||
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||
|
||||
## How the pieces fit
|
||||
|
||||
```
|
||||
keyboard/mouse driver, input-source input service subscriber(s)
|
||||
----------------------------------- ------------- -------------
|
||||
connectSource(); loop: replyWait: subscribeKeyboard()/…All:
|
||||
publishKeyboardEvent(k) ─ ipc_call ─▶ publish → broadcast: createIpcEndpoint()
|
||||
publishMouseEvent(m) for each sub whose callCap(subscribe,
|
||||
publishJoystickEvent(j) mask matches event.device: send_cap = ep,
|
||||
ipc_send(sub_ep) ──────▶ device_mask)
|
||||
reply ok loop: next()
|
||||
subscribe → store {ep cap, └─ replyWait(ep)
|
||||
task id, device_mask} → InputEvent
|
||||
```
|
||||
|
||||
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||
- A **source** (a keyboard, mouse, or joystick driver) calls `input.connectSource()` and the
|
||||
method for its class: `publishKeyboardEvent`, `publishMouseEvent`, or
|
||||
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||
subscriber.
|
||||
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||
|
||||
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||
the service delivers to its endpoint, which only the same thread could receive).
|
||||
|
||||
## Status and follow-ups
|
||||
|
||||
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
interrupt, drains port 0x60, routing every byte by the status register's
|
||||
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||
a future settings source.
|
||||
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||
its one endpoint, acking whichever line the notification's badge names.
|
||||
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
assembles the forwarded bytes with
|
||||
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||
`scroll` events. The hardware-free `input-source` still rotates through all three classes
|
||||
synthetically (including a joystick, which has no driver yet) via the
|
||||
`input.synthetic*Event` helpers.
|
||||
- **Drop-oldest under overflow** is a defined loss; the 16-slot ring absorbs normal bursts.
|
||||
Real backpressure/flow-control is future work.
|
||||
- **`publish` is unauthenticated** — any process may publish, consistent with the current
|
||||
bring-up trust model (see [driver-model.md](driver-model.md)). A source capability is
|
||||
future work.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `input` case (`python3 test/qemu_test.py input`, in
|
||||
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||
exercising `ipc_send`, capability-passing subscription, and per-device routing. Each
|
||||
serial line names the class received, so the log shows all three arriving on one stream.
|
||||
|
||||
## See also
|
||||
|
||||
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||
@@ -0,0 +1,556 @@
|
||||
# Native Intel iGPU display support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *minimal, display-only*
|
||||
native driver for an **Intel integrated GPU** — EDID read + mode-set + framebuffer scanout, with
|
||||
**no** 3D/media/compute — would take, and how it slots into danos's pluggable scanout
|
||||
architecture. It is a survey of primary sources (Intel's open-source
|
||||
[Programmer's Reference Manuals](https://www.intel.com/content/www/us/en/docs/graphics-for-linux/developer-reference/1-0/overview.html),
|
||||
coreboot's [libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html), the Linux
|
||||
[i915 display](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/i915/display) driver,
|
||||
and Haiku's [intel_extreme](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)),
|
||||
not an implementation. It is the companion to [nvidia-gpus.md](nvidia-gpus.md) and should be read
|
||||
against it — the two answer the same question for opposite silicon.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the v2
|
||||
model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- **Intel is a materially easier, lower-tier target than the NVIDIA RTX 3060 — and the reason is
|
||||
documentation, not silicon.** Intel publishes official, register-level, per-platform **Display
|
||||
Engine** PRMs with named registers, bitfields, and numbered enable sequences; NVIDIA publishes
|
||||
no display PRM and forces reverse-engineering against GPL nouveau. A minimal Intel display-only
|
||||
driver is roughly **tier 2 to low-tier 3** for well-covered generations (Skylake / Kaby Lake /
|
||||
Coffee Lake), versus NVIDIA's **tier 4** for GA106. This is the load-bearing conclusion.
|
||||
- **The display block is a genuinely separable register domain.** Mode-set + scanout touch only
|
||||
display registers (pipes, planes, transcoders, DDI buffers, PLLs, power wells, GMBUS/AUX) — **no
|
||||
render engine, no command streamer, no GEM/3D, no signed microcode.** Two small carve-outs, both
|
||||
trivial pokes that do *not* pull in the render engine: a real CDCLK frequency change writes the
|
||||
shared GT PCODE mailbox, and the plane's surface register is a GGTT (memory-interface) address.
|
||||
- **There is no firmware wall on the display path.** The only display microcontroller (DMC / "CSR",
|
||||
Skylake+) is **optional** — its sole job is saving/restoring display state across DC5/DC6
|
||||
low-power idle. Without it, i915 prints "Disabling runtime power management" and mode-sets and
|
||||
scans out normally. GuC/HuC are render/media coprocessors, never touched by a display driver.
|
||||
Pre-Skylake parts have no display microcontroller at all yet mode-set fine. There is **nothing
|
||||
analogous to NVIDIA's GSP**.
|
||||
- **The scanout memory model is dramatically simpler than a discrete GPU.** Intel iGPUs have **no
|
||||
VRAM**: the display scans out of ordinary system RAM addressed through the Global GTT (GGTT), a
|
||||
flat single-level page table. Linear (untiled) framebuffers are first-class. You need **no
|
||||
GEM/TTM, no VMM, no VRAM allocator, no BAR1 aperture juggling** — the exact machinery the NVIDIA
|
||||
path forces on you.
|
||||
- **coreboot libgfxinit is a compact, complete, display-only reference** doing precisely this scope
|
||||
(EDID + PLL/mode-set + scanout, zero 3D) in ~22k lines of formally-analysed SPARK/Ada — versus
|
||||
i915's ~400k lines. It is a *read-and-reimplement* reference, not drop-in code (GPL-2.0-or-later,
|
||||
and Ada, not Zig).
|
||||
- **The clean-room, permissively-licensed path is real** — you can implement from the PRM without
|
||||
reading GPL code, and Haiku's MIT `intel_extreme` is a permissive precedent. This is the decisive
|
||||
contrast with NVIDIA, where no vendor register spec exists.
|
||||
- **The practical catch is hardware, not software.** On a desktop with an RTX 3060, the monitor is
|
||||
almost certainly cabled to the *card*, so an iGPU driver would light a dark motherboard port; the
|
||||
CPU may be an **F-SKU with the iGPU fused off entirely**; and every clean-room reference targets
|
||||
*older* Intel. Intel is the right target to **learn** display bring-up — "run it on my machine"
|
||||
is a separate, machine-dependent question that may not resolve in the reader's favour.
|
||||
- **Recommendation:** as with the NVIDIA doc, GOP already gives native-resolution scanout with zero
|
||||
GPU code. A native Intel driver buys runtime mode changes, hardware vsync, and multihead — and it
|
||||
reaches "first pixel" far faster than the NVIDIA path *if* the target machine actually has a
|
||||
usable, cable-attached iGPU of a documented generation.
|
||||
|
||||
## Display engine architecture, and why it's separable
|
||||
|
||||
For the common single-display path (SST DisplayPort / HDMI / eDP), the Intel display data flow is a
|
||||
small, fully documented, essentially fixed sequence:
|
||||
|
||||
```
|
||||
memory surface → PLANE(s) → PIPE → TRANSCODER → DDI (drives IO/PHY) → connector
|
||||
```
|
||||
|
||||
The Tiger Lake PRM Vol 12 states it verbatim: *"The front end of the display contains the pipes.
|
||||
The pipes connect to the transcoders. The transcoders, except for wireless, connect to the DDIs to
|
||||
drive the IO/PHY."* A **pipe** blends planes (primary/sprite/cursor) into one raster stream; the
|
||||
**transcoder** wraps it in port-protocol timing (DP/HDMI/eDP/DSI); the **DDI** is the physical port
|
||||
and PHY. Pipe, Planes, Transcoder, and Digital Display Interface are each first-class PRM chapters
|
||||
with per-object files in libgfxinit
|
||||
([TGL PRM Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||
|
||||
**Two honest qualifications** the raw research overstated (per verification):
|
||||
|
||||
- The pipeline is *not* strictly linear in all cases — the same PRM pages document optional branches
|
||||
a minimal driver simply ignores (wireless writeback to memory, MIPI DSI, DisplayPort multistream
|
||||
many-to-one, DSC/tiled pipe-joining). Ignoring them does not weaken feasibility.
|
||||
- The four-object model *as named* is **Haswell-onward** (DDI introduced ~2013), not "every gen."
|
||||
Pre-Haswell used FDI + PCH transcoders + port-specific encoders. Within the modern iGPU range
|
||||
danos would realistically target (Skylake → Meteor/Lunar Lake) the model is stable.
|
||||
|
||||
**The DPLL/clock block is a separate, per-port programmable clock source** and is one of the harder,
|
||||
most gen-specific pieces: pick/enable a PLL, route its output to the DDI, then bring up the port.
|
||||
The register layout and divider math change substantially per generation — pre-SKL SPLL/WRPLL/LCPLL,
|
||||
Skylake+ shared DPLL0–3, Gen11+ combo-PHY plus Type-C MG/DKL PLLs. Pixel-clock computation is a
|
||||
classic per-gen rewrite.
|
||||
|
||||
### Separable from render — the single most important enabler
|
||||
|
||||
The display is a distinct register domain from render/media, and this is confirmed at the primary
|
||||
level: the TGL PRM ships display as its own volume (Vol 12), separate from Render Engine (Vol 9) and
|
||||
Media (Vol 11); Linux's KMS "is provided by Intel Display Driver, and **shared with drm/xe**"
|
||||
([kernel.org i915](https://docs.kernel.org/gpu/i915.html)) — i.e. the display module is
|
||||
reused across two different GPU drivers. A full mode-set lights a display end-to-end using only power
|
||||
wells, PLL/port-clock, DDI-buffer/PHY, transcoder and pipe registers — **zero render commands, zero
|
||||
GEM objects, zero command-streamer.** libgfxinit is decisive proof: complete EDID + modeset +
|
||||
framebuffer with no render/3D code at all.
|
||||
|
||||
Two carve-outs the "touches ONLY display registers" phrasing needs (per verification), **neither of
|
||||
which drags in the render engine**:
|
||||
|
||||
1. A mode-set that changes the **Core Display Clock (CDCLK)** frequency/voltage pokes the shared **GT
|
||||
Driver Mailbox** (PCODE/PCU power-controller interface), per Vol 12's own "Display Voltage
|
||||
Frequency Switching" step. A trivial register handshake, documented alongside the display sequence.
|
||||
2. The primary plane's surface register (`PLANE_SURF`) holds a **GGTT graphics address** (a
|
||||
memory-interface concept, not covered in Vol 12). Using pre-mapped stolen memory — as libgfxinit
|
||||
does — sidesteps any active GGTT programming. See [Memory and scanout](#memory-and-scanout).
|
||||
|
||||
### Per-gen churn: what's stable, what you rewrite
|
||||
|
||||
The **object model** (pipes/planes/transcoders/DDIs, GMBUS-for-EDID, double-buffered plane registers
|
||||
armed atomically) is conceptually stable from Ironlake/Haswell through Tiger Lake. What you rewrite
|
||||
per generation is:
|
||||
|
||||
1. the **CPU-vs-PCH split and interconnect**,
|
||||
2. the **port/PHY + DPLL** programming,
|
||||
3. **register offsets + power-well / CDCLK topology**, and
|
||||
4. the **mode-set enable sequence itself** (power-well ordering, PLL lock, DDI-buffer enable,
|
||||
transcoder clock-select) — an effective fourth axis the raw research folded into (1)/(2).
|
||||
|
||||
Interconnect eras, with the timeline **corrected** (the cited Haiku doc was chronologically loose):
|
||||
|
||||
- **Gen5 Ironlake (2010) → Ivy Bridge:** FDI (Flexible Display Interface) links the CPU display
|
||||
engine to PCH-resident ports. The FDI/PCH-split era begins at **Ironlake**, not Gen7.
|
||||
- **Haswell (Gen7.5):** the main digital outputs come **back onto the CPU die as DDIs** (DDI A = eDP)
|
||||
— the *opposite* of "moving output to the PCH," and it collapses the FDI/PCH dance **for the
|
||||
digital ports only**. FDI is **retained** for the legacy VGA/CRT path (DDI E → PCH CRT DAC), so a
|
||||
driver gets the single DDI code path only by omitting analog VGA (which a minimal driver does).
|
||||
- **Skylake (Gen9):** reworks clock/PLL, CDCLK, and the power-well model; introduces the optional DMC.
|
||||
- **Gen11 Ice Lake / Gen12 Tiger Lake:** add combo-PHY + USB-Type-C/Thunderbolt MG/DKL PHYs — the
|
||||
single biggest cost increase, and the reason "newest silicon" is *not* the easiest target. (DSC is
|
||||
documented per-**pipe**; MSO is an eDP feature — not "per-transcoder" as the raw research said.)
|
||||
|
||||
### The tractable sweet spot
|
||||
|
||||
The documented, tractable sweet spot for a from-scratch display-only driver is the
|
||||
**Haswell (Gen7.5) / Broadwell (Gen8) DDI family, with Skylake (Gen9) as the modern-hardware pick**
|
||||
since it shares the same DDI object model. Rationale:
|
||||
|
||||
- Broadwell has a complete, freely downloadable
|
||||
[PRM Vol 11 Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf);
|
||||
its engine (3 pipes A/B/C, 4 transcoders incl. transcoder-EDP that floats onto any pipe, DDI A–E,
|
||||
WRPLL/SPLL/LCPLL) is the classic "DDI + transcoder + WRPLL" model.
|
||||
- It predates the combo-PHY / Type-C / MG-DKL complexity of Ice Lake / Tiger Lake.
|
||||
- libgfxinit's DDI **connector/EDID/DP layer is uniform from Haswell through Coffee Lake**, so the
|
||||
hardest-to-get-right port logic generalises widely.
|
||||
|
||||
Two supporting claims from the raw research are **wrong and corrected here (verification):**
|
||||
|
||||
- **The BDW and SKL PRMs are NOT 0BSD-licensed.** Both carry a Creative Commons
|
||||
**Attribution-NoDerivatives** notice. Only the *newer* OSRC PRMs (Tiger Lake 2021 onward) put their
|
||||
embedded code samples under **Zero-Clause BSD**. So for the recommended Haswell/Broadwell/Skylake
|
||||
generations there are no "copy-pasteable 0BSD code samples" — the legal basis is *reimplementation
|
||||
from a CC-BY-ND spec* (register facts are not copyrightable), not copying.
|
||||
- **FDI+PCH is not fully eliminated on Haswell/Broadwell.** The BDW PRM keeps FDI for the DDI E → PCH
|
||||
CRT DAC. The "one DDI code path" holds only for the digital outputs a minimal driver targets.
|
||||
|
||||
Sandy/Ivy Bridge (Gen6/7) is where the hobby-doc walkthroughs concentrate (the OSDev GMBUS/EDID
|
||||
material) but carries the FDI+PCH split cost. *(Low confidence on the OSDev specifics — the wiki
|
||||
returns 403 to automated fetches and its "guaranteed to work" phrasing is a hobby assertion, not a
|
||||
silicon guarantee.)*
|
||||
|
||||
## Documentation — and the clean-room question
|
||||
|
||||
This is the crux of the whole comparison. **Intel hands you the register spec that NVIDIA withholds.**
|
||||
|
||||
- The Tiger Lake **"Vol 12: Display Engine"** PRM is a real, first-party, open-source document —
|
||||
**433 pages, verified by direct download** — with named registers + addresses + bitfield tables
|
||||
(`TRANS_DDI_FUNC_CTL`, `DDI_BUF_CTL`, `DP_TP_CTL`, `PLANE_STRIDE`, `DPLL_CFGCR0/1`, `CDCLK_CTL`,
|
||||
`PWR_WELL_CTL_DDI`, …) and **numbered, step-by-step enable sequences** with explicit writes, wait
|
||||
conditions, and microsecond timeouts. It even includes the "magic value" tables older PRMs deferred
|
||||
to the driver (DisplayPort PLL DCO/divider values; voltage-swing/de-emphasis in mV). *"A spec you
|
||||
could write a driver from directly"* is well-supported, not hyperbole
|
||||
([TGL Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||
- **Clean-room, permissively-licensed implementation is legally and practically feasible from the
|
||||
PRM alone.** CC-BY-ND governs redistribution of the *document*; register addresses and bit
|
||||
definitions are functional facts, and original code implementing a described hardware interface is
|
||||
not a derivative of the PDF. *(This is standard copyright reasoning, not adjudicated case law —
|
||||
treat it as well-grounded, not settled.)* Two independent implementations already exist built
|
||||
essentially from these docs (libgfxinit, Haiku), so the spec is demonstrably sufficient.
|
||||
|
||||
**The documentation ceiling — corrected.** The raw research said public PRMs stop "roughly at Ice
|
||||
Lake / Tiger Lake." Verification refuted this: full public **"Vol 12 Display Engine"** PRMs exist for
|
||||
Ice Lake, Lakefield, Tiger Lake, Rocket Lake, DG1, **and DG2/Arc "Alchemist" (Gen12.5, 2022)** —
|
||||
[the ACM display PRM is public](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf).
|
||||
The genuine cliff is **Meteor Lake (2023) and newer**: those have only a high-level architecture
|
||||
overview, no register-level display PRM, and i915 references their display registers by opaque
|
||||
internal **Bspec numeric IDs**. Alder Lake and Raptor Lake iGPUs are Gen12 Xe-LP display — the same
|
||||
IP as Tiger Lake — so despite lacking a dedicated PRM they are effectively covered by the TGL PRM.
|
||||
|
||||
Net: a from-docs driver can confidently target **Skylake through DG2/Arc**, which is essentially the
|
||||
entire current laptop/NUC installed base; only Meteor Lake and later slide back toward the NVIDIA
|
||||
situation (reverse-engineering or reading GPL i915). The PRMs also survived 01.org's shutdown and are
|
||||
mirrored in several stable places (Intel's cdrdv2 host, the
|
||||
[Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) for Gen4–Gen9.5,
|
||||
[kiwitree](https://kiwitree.net/~lina/intel-gfx-docs/prm/), x.org) — not a single point of failure.
|
||||
*(Note: the Igalia archive stops at Kaby Lake and contains no Display Engine volume; the TGL/DG2
|
||||
display PRMs are separate Intel/x.org downloads.)*
|
||||
|
||||
## coreboot libgfxinit — the native reference
|
||||
|
||||
[libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html) is the closest thing to a template danos
|
||||
could ask for: a self-contained **native modeset library** (no VBIOS/int10, no firmware blobs) that
|
||||
probes displays via EDID over DDC/I²C and DP AUX, and drives LVDS, eDP, DP1–3, HDMI1–3, analog VGA,
|
||||
plus USB-C DP/HDMI alt-mode on Tiger Lake. It sets up pipes (Primary/Secondary/Tertiary), planes,
|
||||
transcoders, PLLs, panel power/backlight, the GTT, and framebuffer scanout — **display-only, zero
|
||||
3D/media/compute**, which is exactly danos's scope. Its public entry is essentially
|
||||
`Initialize()` then `Update_Outputs(Pipe_Configs)`, where each `Pipe_Config` carries
|
||||
`{Port, Framebuffer, Cursor, Mode}` — a near-perfect fit for a pluggable scanout backend.
|
||||
|
||||
Why it beats i915 as a reference (**verified by measurement**): **131 Ada source files, ~818 KB,
|
||||
~22k code lines** across *all* generations, factored precisely along the axes you care about (`edid`,
|
||||
`dp_aux`, `dp_training`, `pipe_setup`, `transcoder`, `plls`, `connectors`, `port_detect`), with
|
||||
**none** of the DRM/KMS/GEM/TTM, GT/3D, RC6/RPS, or GuC/HuC machinery that makes
|
||||
`drivers/gpu/drm/i915` **~419k lines / 900 files / 12 MB**. (A grep confirms *zero* gem/ttm/guc/huc/
|
||||
execbuf identifiers in the tree.) It depends only on a small HW-access shim, `libhwbase`
|
||||
(`HW.PCI`, `HW.Port_IO`, `HW.MMIO`, `HW.Time`), which maps naturally onto danos's MMIO-grant + IPC
|
||||
primitives — you provide Zig equivalents and the modeset logic sits on top. *(Correction to the raw
|
||||
research: the widely-quoted "~13–14k LOC" is only the generic `common/` layer; the eight
|
||||
per-generation subdirs roughly double it.)*
|
||||
|
||||
**It is a read-and-reimplement reference, not drop-in code.** Two hard constraints:
|
||||
|
||||
- **License is GPL-2.0-or-later** (the COPYING file is GPLv2; per-file headers add "or any later
|
||||
version"). The CC-BY-4.0 on the docs *site* is a footer, not the source license. Copyleft applies
|
||||
to ported code.
|
||||
- **It is SPARK/Ada, and designed to run as coreboot boot-firmware**, not a runtime OS driver. A
|
||||
danos port means either an Ada/GNAT toolchain in the build or hand-transliteration into Zig; the
|
||||
SPARK "absence of runtime errors" proof does **not** carry over to your reimplementation (and note
|
||||
it proves absence of runtime errors, **not** functional modeset correctness).
|
||||
|
||||
Two more caveats worth knowing: its **error handling is limited** — "only the case that no display
|
||||
could be found counts as failure"; a later DP link-training failure is *not* propagated. And its
|
||||
**verified-in-coreboot** hardware list stops at **Coffee Lake + Apollo Lake**, even though the tree
|
||||
contains a `tigerlake/` directory (Ice Lake has no directory at all, and Alder Lake support is only
|
||||
"begun"). So treat Haswell..Coffee Lake as the trustworthy transliteration window and TGL as
|
||||
present-but-less-proven.
|
||||
|
||||
The orchestration reads as a clean state machine (`hw-gfx-gma.adb` `Enable_Output`):
|
||||
`Fill_Port_Config → Preferred_Link_Setting → PLLs.Alloc → [retry] Connectors.Pre_On →
|
||||
Display_Controller.On → Connectors.Post_On`, with a literal *"try each DP-lane configuration twice"*
|
||||
inner retry and an outer link-setting step-down. `hw-gfx-dp_training.adb` (398 lines) is a complete,
|
||||
generic DP link-training implementation (TP1/TP2/TP3, CR + EQ loops, swing/pre-emphasis adjust from
|
||||
sink status). Per-generation buffer translations plug in underneath via
|
||||
`Program_Buffer_Translations`, gated on `Config.Has_DDI_Buffer_Trans`. All of this was confirmed
|
||||
against the source line-by-line.
|
||||
|
||||
## The EDID + mode-set path (Haswell/Broadwell target)
|
||||
|
||||
The whole path is memory-mapped register programming with polled status bits — no command ring, no
|
||||
microcode, no DMA channel.
|
||||
|
||||
**EDID over DDC (GMBUS).** Pure MMIO poking of the GMBUS I²C controller (`GMBUS0`–`GMBUS5`): `GMBUS0`
|
||||
selects pin-pair/port + clock; `GMBUS1` carries slave address (`0x50` for EDID), byte count,
|
||||
direction, SW-ready; `GMBUS2` exposes HW-ready/NAK/ACTIVE to poll; `GMBUS3` is a 4-byte data FIFO;
|
||||
`GMBUS5` gives the 2-byte segment index for E-DDC. A read is: write `GMBUS0`, write `GMBUS1`
|
||||
(`CYCLE_WAIT | count | SLAVE_READ | SW_RDY | slave<<addr`), loop {poll `HW_RDY`, read 4 bytes}, then
|
||||
STOP ([i915 intel_gmbus.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_gmbus.c)).
|
||||
|
||||
**EDID + DPCD over DP AUX.** For DisplayPort/eDP, EDID (as I²C-over-AUX to `0x50`) and all DPCD
|
||||
capability/link-status registers are read over the AUX channel: per-DDI `DDI_AUX_CTL` + 5×
|
||||
`DDI_AUX_DATA`. Build a 3–5 byte header + payload, set SEND_BUSY, poll it clear, read
|
||||
DONE/TIMEOUT/RECEIVE_ERROR. Message size 1–20 bytes; spec requires ≥3 retries. On Haswell/BDW the AUX
|
||||
clock divider is programmed explicitly; SKL+ derive it automatically
|
||||
([i915 intel_dp_aux.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dp_aux.c)).
|
||||
Both GMBUS and DP-AUX live in libgfxinit's shared `common/` — cheap and nearly gen-invariant.
|
||||
|
||||
**The mode-set is a fixed, documented register sequence.** The Broadwell DisplayPort enable order
|
||||
(verbatim from BDW PRM Vol 11, pp.98–99): (1) DDI lane capability; (2) panel power sequencing if
|
||||
needed; (3) enable the CPU display PLL (WRPLL/SPLL) and wait ~20 µs; (4) Port Clock Select → DDI,
|
||||
enable `DP_TP_CTL` with training pattern 1, configure `DDI_BUF_TRANS`, enable `DDI_BUF_CTL`, wait
|
||||
>518 µs, run link training, set `DP_TP_CTL` to Normal (Idle first for eDP); (5) Transcoder Clock
|
||||
Select, enable the plane, panel fitter if needed, program transcoder timings + M/N/TU, enable
|
||||
`TRANS_DDI_FUNC_CTL`, enable `TRANS_CONF`, then backlight. Disable is the exact reverse — a bounded
|
||||
checklist.
|
||||
|
||||
**DisplayPort/eDP link training is driver-driven in software over AUX** — the CPU runs the
|
||||
clock-recovery and channel-equalization state machines by hand; it is **not** offloaded to a hardware
|
||||
sequencer or firmware. The source side exposes only primitives: `DP_TP_CTL` selects the training
|
||||
pattern the port emits; `DDI_BUF_CTL`/`DDI_BUF_TRANS` set voltage-swing/pre-emphasis. The driver
|
||||
loops: emit pattern + set source levels → write `TRAINING_PATTERN_SET` (DPCD 0x102) + `TRAINING_LANEx_SET`
|
||||
(0x103) over AUX → delay (100 µs CR / 400 µs EQ) → read `LANE_STATUS` → on failure adjust to the
|
||||
sink's `ADJUST_REQUEST` values and retry. A few hundred lines of ordinary CPU/AUX code (libgfxinit
|
||||
`Train_DP`: CR loop 1..32, EQ loop 1..6). **This is the single fiddliest, most fragile piece** — a
|
||||
TMDS/HDMI panel avoids it entirely, and targeting an already-lit eDP panel avoids most of it.
|
||||
|
||||
**The clock (WRPLL) is documented divider math, not a magic table.** On Haswell/BDW the WRPLL derives
|
||||
the symbol clock from a 2700 MHz LCPLL reference through R2/N2/P dividers with VCO 2400–4800 MHz —
|
||||
small integer arithmetic. DP is *easier* than HDMI because it runs at a few fixed link rates (1.62 /
|
||||
2.7 / 5.4 GHz), so a DP/eDP-only minimal driver can often use fixed rates and skip most of the search.
|
||||
|
||||
**Plane/scanout programming is trivial for a compositor.** The primary plane is `PRI_CTL`
|
||||
(enable + pixel format), `PRI_STRIDE`, `PRI_SURF` (surface base — writing it triggers the atomic
|
||||
update), `PRI_OFFSET`; formats include 32-bit BGRX 8:8:8 and 16-bit BGRX 5:6:5 — a direct match for a
|
||||
linear XRGB compositor buffer. Plane registers are double-buffered and latch at vblank via an
|
||||
**arming** write — so a page-flip is "write base + stride + size, then the arming write." This is
|
||||
*exactly* the primitive danos's damage-driven compositor already expresses over GOP/virtio-gpu; the
|
||||
incremental work is "program these display-domain registers," not a new scanout model. The panel
|
||||
fitter (`PF_WIN_POS`/`PF_WIN_SZ`/`PF_CTRL`) can be left disabled for native-resolution scanout;
|
||||
Skylake+ replaces it with a shared pipe-scaler (`PS_CTRL`).
|
||||
|
||||
**Smallest useful target:** eDP (DDI A / transcoder-EDP) or a single DP output at native resolution,
|
||||
panel fitter off, plane in 32bpp XRGB. That is: GMBUS + I²C-over-AUX EDID/DPCD, one fixed-rate or
|
||||
WRPLL config, the ~20-step enable sequence, the software CR/EQ loop, and `PRI_*` plane setup with
|
||||
`PRI_SURF`-write flips. Out of scope: 3D, media, tiling, RC6/power-gating, PSR, audio.
|
||||
|
||||
## Memory and scanout
|
||||
|
||||
This is where Intel's *architecture* — not just its docs — makes the job smaller, and it is the
|
||||
biggest single simplification versus a discrete GPU.
|
||||
|
||||
- **No VRAM.** Intel iGPUs have a unified memory architecture; the display scans out of ordinary
|
||||
**system RAM** addressed through the **Global GTT (GGTT)**. The only way to give the GPU memory is
|
||||
to bind system pages into the GGTT
|
||||
([i915/GEM crashcourse](https://blog.ffwll.ch/2012/10/i915gem-crashcourse.html)).
|
||||
- **The plane surface register is a GGTT offset**, not a raw physical address — the display walks the
|
||||
GGTT to fetch pixels, so a scanout buffer must be GGTT-mapped (global, not per-process). libgfxinit
|
||||
writes the framebuffer offset straight into `DSPSURF`/`PLANE_SURF` masked to 4 KB.
|
||||
- **Linear (untiled) scanout is a first-class supported mode** — the plane's tiling field value 0 is
|
||||
Linear. No X/Y/Yf tiling engine is needed for a display-only driver. (UEFI GOP itself hands off a
|
||||
linear framebuffer the plane is already scanning.)
|
||||
- **No memory manager.** You need only (1) some contiguous-ish system pages and (2) GGTT PTEs
|
||||
pointing at them (`physical_addr | valid_bit` — the GGTT is a flat single-level array of PTEs in
|
||||
the `GTTMMADR` MMIO BAR), then program the plane. **No GEM/TTM/PPGTT/GuC.** coreboot's native-init
|
||||
literally does `for(i…) WRITE32(base + i*inc | 1, (i*4) | 1)`.
|
||||
- **"Stolen memory"** (GSM/DSM) is firmware-reserved system RAM where the firmware places the GGTT
|
||||
itself and the boot framebuffer. A driver is not obligated to keep scanout there — it can rebind
|
||||
GGTT entries to its own pages. Stolen memory matters mainly for *inheriting* the GOP framebuffer at
|
||||
handoff.
|
||||
|
||||
**The contrast with NVIDIA is stark.** On a discrete GPU the scanout surface must live in **VRAM**
|
||||
(nouveau always pins scanout to VRAM), CPU access goes through the **BAR1** aperture (which on
|
||||
consumer cards can be far smaller than total VRAM unless Resizable BAR is on), and you need a
|
||||
contiguous aligned VRAM allocator plus a BAR1 mapping. The Intel iGPU path **eliminates all of that**
|
||||
— scanout is plain system RAM, and a userspace compositor can write the framebuffer pages directly
|
||||
(as danos already does with the GOP WC framebuffer).
|
||||
|
||||
Because danos boots via GOP, an Intel driver attaches to a display whose **GGTT is already populated
|
||||
and whose plane is already scanning a linear framebuffer at native resolution.** A minimal driver can
|
||||
reuse that live mapping and reprogram the running plane rather than come up from cold — the same
|
||||
"attach to a live display" advantage the NVIDIA doc identifies, but with a far smaller register
|
||||
surface and no firmware wall. *(Low-confidence, per-target details to pin from the specific gen's
|
||||
PRM: GGTT PTE size — 4-byte pre-gen8 vs 8-byte gen8+ — the `GTTMMADR`/aperture BAR layout, surface
|
||||
alignment — 4 KB floor but some gens/tilings want 256 KB — and whether the display's GGTT-mediated
|
||||
DMA sits before or after danos's M16 IOMMU on the target platform.)*
|
||||
|
||||
## Firmware
|
||||
|
||||
A minimal display-only Intel driver is **effectively firmware-free — more so than NVIDIA.**
|
||||
|
||||
- **DMC (Display Microcontroller, "CSR", Skylake+) is NOT required for mode-set or scanout.** Its
|
||||
sole job is saving/restoring display-engine registers across DC5/DC6 low-power idle. Absent, i915
|
||||
prints *"Failed to load DMC firmware … Disabling runtime power management"* and the display
|
||||
mode-sets and scans out normally — you lose only the deep display idle states, not output
|
||||
([intel_dmc.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dmc.c);
|
||||
corroborated by multiple distro bug threads). *(A source-level `HAS_DMC` early-return citation would
|
||||
strengthen this beyond distro testimony, but the conclusion is well-supported.)*
|
||||
- **Pre-Skylake parts have no display microcontroller at all** yet perform full mode-set (and even
|
||||
Panel Self Refresh). This confirms the display engine is fundamentally CPU/MMIO-driven; the
|
||||
microcontroller is an add-on for autonomous idling, not a prerequisite for lighting a panel.
|
||||
Targeting a pre-Skylake or DMC-optional generation sidesteps the question entirely.
|
||||
- **GuC and HuC are render/media microcontrollers on the GT side** — GuC schedules the render engines,
|
||||
HuC assists HEVC/H.265 codec (plus later HDCP/PXP/GSC). Neither is in the scanout path; a
|
||||
display-only driver never loads them
|
||||
([kernel.org microcontrollers](https://docs.kernel.org/gpu/i915.html)).
|
||||
- **PSR firmware lives on the panel**, not in the OS — a minimal driver simply doesn't enable PSR.
|
||||
- **Type-C/TCSS (Ice Lake+) firmware** (PMC/IOM/PHY) is part of platform BIOS/coreboot init and the
|
||||
hardware, *not* a signed blob the display driver loads at runtime. A driver attaching to an
|
||||
already-lit GOP connector, or targeting classic DDI ports, avoids it. *(Cold DP-alt-mode changes
|
||||
from a userspace driver on modern TCSS platforms were not traced to primary source — flagged.)*
|
||||
|
||||
There is **no signed-firmware wall over the Intel GPU at all** on the display path. This is the
|
||||
architectural opposite of NVIDIA's mandatory, unsignable, ABI-unstable GSP — which even on the
|
||||
near-side "direct" display path is a permanent maintenance liability for anything beyond scanout.
|
||||
|
||||
## Licensing
|
||||
|
||||
The situation is *better* than NVIDIA's but still nuanced.
|
||||
|
||||
- **The two best code references are both GPL** — Linux i915 (GPL-2.0) and coreboot libgfxinit
|
||||
(GPL-2.0-or-later). You cannot copy either into a permissively-licensed danos. libgfxinit's WRPLL
|
||||
divider math is itself copied from i915, so it carries the same encumbrance.
|
||||
- **But you don't need to copy code.** The Intel PRM is a *specification*, and a clean-room Zig
|
||||
implementation written from the PRM (using libgfxinit/i915 only to understand behaviour, never to
|
||||
copy) is legitimate — register numbers and bit definitions are functional facts, not copyrightable
|
||||
expression. This is the exact inverse of the NVIDIA case, where no such spec exists and the only
|
||||
guide is the GPL/RE'd code itself.
|
||||
- **A permissive precedent exists: Haiku's `intel_extreme` is MIT-licensed** and was built from
|
||||
Intel's public docs. So if danos wants a permissive license, the model is: implement from the PRM,
|
||||
optionally read MIT Haiku for structure, treat GPL libgfxinit/i915 as documentation-of-last-resort.
|
||||
- **A licensing nuance on the recommended generations:** the "copy the 0BSD PRM code samples" shortcut
|
||||
only applies to Tiger-Lake-era (2021+) PRMs. The Haswell/Broadwell/Skylake PRMs are CC-BY-ND, so
|
||||
their register *facts* are free to implement but there are no code samples to lift.
|
||||
|
||||
As with the NVIDIA doc: danos's userspace-driver-over-IPC model (a driver is a separate process behind
|
||||
a defined protocol) is the cleanest possible license boundary if the project ever chooses to ship a
|
||||
GPL display-driver binary and keep the rest of danos permissive — but that is a boundary judgement
|
||||
wanting real diligence, not a settled fact. The clean-room-from-PRM route avoids the question.
|
||||
|
||||
## Prior art outside Linux
|
||||
|
||||
This is a **real contrast with NVIDIA**, where no one has built a from-scratch native driver outside
|
||||
Linux. For Intel there are **multiple independent, non-Linux, clean-room native modeset
|
||||
implementations** to learn from:
|
||||
|
||||
- **coreboot libgfxinit** — SPARK/Ada, G45/GM45 and Arrandale → Coffee Lake + Apollo Lake (TGL
|
||||
in-tree), the strongest structural reference.
|
||||
- **Haiku `intel_extreme`** — modeset-only (no 2D/3D accel), **MIT-licensed**, i845 through Sandy
|
||||
Bridge solid, newer Gemini/Ice/Tiger Lake in progress but "hit or miss, as the driver lags behind
|
||||
the specs" ([Haiku generations](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html),
|
||||
[Phoronix Sept 2024](https://www.phoronix.com/news/Haiku-OS-September-2024)).
|
||||
- **SerenityOS** — added basic native Intel graphics ([PR #6277](https://github.com/SerenityOS/serenity/pull/6277)),
|
||||
though only for very old ICH7-class hardware.
|
||||
- **managarm** — native Intel G45 support.
|
||||
|
||||
The catch: **every clean-room non-Linux implementation targets old hardware.** A modern Gen12 "Xe"
|
||||
desktop iGPU is beyond all of them; for the very newest parts only GPL i915 covers the registers. So
|
||||
the wealth of prior art is real but concentrated below Tiger Lake.
|
||||
|
||||
## The practical desktop caveat
|
||||
|
||||
Before any effort estimate is trusted, three hardware realities — the honest reason "Intel is easier"
|
||||
does **not** automatically mean "it'll light up the reader's monitor":
|
||||
|
||||
1. **Muxing / cabling.** On a desktop with a discrete RTX 3060, the monitor is almost certainly
|
||||
plugged into the *card's* outputs, not the motherboard's. An iGPU driver would light a
|
||||
**different, currently-dark** output. To see danos on Intel the reader would have to physically
|
||||
move the cable to a motherboard video port **and** likely enable the iGPU / "IGD Multi-Monitor" in
|
||||
BIOS. Intel-first probably does **not** light the current display without re-cabling.
|
||||
2. **No iGPU at all.** Intel **F-SKU** desktop chips (i5-9400F, i5-12400F, i5-13400F, i7-13700KF, …)
|
||||
ship the graphics **fused off** and cannot be re-enabled. These are extremely common in
|
||||
budget/mid gaming builds paired with an RTX 3060. On an F-SKU (or an X-series HEDT part) the
|
||||
Intel-iGPU path is a **non-starter** regardless of cabling.
|
||||
3. **Generation coverage.** If the CPU *is* a recent non-F part, its iGPU may be Gen12 Xe (Alder/
|
||||
Raptor Lake), beyond libgfxinit's verified set and beyond most non-Linux prior art — leaving GPL
|
||||
i915 (or the TGL-class PRM, which covers Alder/Raptor display IP) as the only reference.
|
||||
|
||||
A cleaner path for *learning* without the hardware lottery: an older bare-metal Intel box (Haswell/
|
||||
Skylake NUC or laptop) whose panel is natively on the iGPU. Note QEMU does **not** emulate an Intel
|
||||
iGPU display engine, so a VM cannot exercise a real Intel modeset path — virtio-gpu (already working)
|
||||
is the VM answer.
|
||||
|
||||
## Alternatives, and the honest Intel-vs-NVIDIA verdict
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; no runtime mode change, no hardware vsync, no multihead |
|
||||
| **Intel iGPU, reuse-GOP** | EDID read + plane page-flips on the GOP-set mode | Still bounded to GOP's resolution; but real driver-owned scanout |
|
||||
| **Intel iGPU, full modeset** (this doc) | Runtime modeset, vsync, multihead, from public docs | Tier 2–3 effort; DP link training; per-gen churn; **needs a cable-attached, documented iGPU** |
|
||||
| **Native NVIDIA GA106 direct** ([nvidia-gpus.md](nvidia-gpus.md)) | Same, on the RTX 3060 the monitor is actually plugged into | **Tier 4**; GPL-only reference; DMA channel modeset; de-emphasised legacy path |
|
||||
| **GA106 via GSP/OGKM** | Also unlocks 3D later | Tier 5; unstable version-pinned firmware ABI |
|
||||
|
||||
**The verdict for *this reader* (RTX 3060 box):** For pure "see danos on my screen," **NVIDIA-direct
|
||||
is paradoxically the more relevant path**, because the monitor is already cabled to the 3060 and GOP
|
||||
already drives it — a native NVIDIA driver reprograms *that* live display. An Intel driver, however
|
||||
much easier to *write*, likely lights a dark motherboard port the reader isn't looking at, or hits an
|
||||
F-SKU with no iGPU.
|
||||
|
||||
**The verdict for *learning display bring-up*:** **Intel wins decisively.** Public register PRMs, four
|
||||
independent open reference drivers, an MIT precedent (Haiku), a compact formally-analysed blueprint
|
||||
(libgfxinit), no signed-firmware wall, no VRAM/BAR memory manager, and a legitimate permissive
|
||||
clean-room path. It reaches "first pixel" far faster than the NVIDIA native path — *on hardware that
|
||||
actually has a cable-attached, documented Intel iGPU.* Those two goals — "run on my machine" and
|
||||
"learn the craft" — point at different silicon, and that is the honest bottom line.
|
||||
|
||||
## "First light" milestones — a danos `.scanout` service
|
||||
|
||||
Framed as a danos `.scanout` service (like the virtio-gpu and proposed NVIDIA ones), inheriting the
|
||||
GOP-initialized display — no firmware, no cold POST:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate the iGPU, map its MMIO BAR (`GTTMMADR` + register block) and the
|
||||
aperture BAR via danos MMIO grants; confirm the display engine is GOP-live.
|
||||
2. **EDID** — implement GMBUS DDC (`0x50`) and DP AUX; read + parse the panel EDID and DPCD caps.
|
||||
*(Smallest self-contained, gen-invariant milestone — a good first commit.)*
|
||||
3. **First pixel = reprogram, don't re-modeset** — with GOP's mode and GGTT mapping inherited,
|
||||
reprogram the running plane (`PRI_CTL`/`PRI_STRIDE`/`PRI_SURF`, linear, 32bpp XRGB) to point at a
|
||||
danos-owned system-RAM buffer; prove a page-flip via the `PRI_SURF` arming write on the *current*
|
||||
mode before changing timings. This defers the entire DPLL/DDI/transcoder/link-training surface —
|
||||
the hardest, most gen-specific ~70% of the work.
|
||||
4. **GGTT ownership** — write your own GGTT PTEs (via an MMIO grant to `GTTMMADR`) pointing at
|
||||
compositor-owned pages, for double-buffered damage-driven present.
|
||||
5. **Wire into the compositor `.scanout` backend** (`attach_scanout`); add vsync via the display
|
||||
vblank interrupt (IRQ-as-IPC).
|
||||
6. **Full mode-set** (the hard, gen-specific step) — for one chosen generation (Haswell/Broadwell or
|
||||
Skylake): WRPLL/DPLL programming, the ~20-step DDI/transcoder/pipe enable sequence, panel power
|
||||
sequencing for eDP (`PP_CONTROL`/`PP_ON_DELAYS`/`PP_OFF_DELAYS` — a common black-screen pitfall).
|
||||
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||
software CR/EQ state machine over AUX. TMDS/HDMI avoids it; a live eDP panel avoids most of it.
|
||||
8. **Multihead**, then optionally a second generation once one is solid.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||
working display, exactly the resilience v2 already provides via re-attach.
|
||||
|
||||
## Reading list
|
||||
|
||||
**Native reference — coreboot libgfxinit (GPL-2.0-or-later, SPARK/Ada):**
|
||||
- `common/hw-gfx-gma.adb` — `Enable_Output`, the end-to-end modeset state machine.
|
||||
- `common/hw-gfx-dp_training.adb` — the complete generic DP link-training CR/EQ loops.
|
||||
- `common/hw-gfx-gma-pipe_setup.adb` — plane/pipe/scaler + `DSPSURF`/`DSPSTRIDE`/`DSPCNTR` scanout.
|
||||
- `common/hw-gfx-gma-transcoder.adb` — timing generator; `common/hw-gfx-edid.adb`,
|
||||
`hw-gfx-gma-i2c.adb`, `hw-gfx-dp_aux_ch.adb` — EDID/DDC/AUX; `hw-gfx-gma-registers.ads` — offsets.
|
||||
- `common/haswell*/`, `skylake/`, `tigerlake/` — the per-gen PLL/PHY/buffer-translation backends.
|
||||
|
||||
**Vendor register specs — Intel OSRC PRMs:**
|
||||
- [Broadwell Vol 11: Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf)
|
||||
(CC-BY-ND) — the recommended Haswell/Broadwell-class enable sequences, plane, panel fitter.
|
||||
- [Tiger Lake Vol 12: Display Engine](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)
|
||||
(code samples 0BSD) — the most complete modern reference incl. PLL/voltage-swing value tables.
|
||||
- [DG2/Arc Vol 12: Display Engine](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf)
|
||||
— the newest public display PRM (Gen12.5, 2022).
|
||||
- [Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) (Gen4–Gen9.5) and the
|
||||
[kiwitree mirror](https://kiwitree.net/~lina/intel-gfx-docs/prm/) — stable mirrors.
|
||||
|
||||
**GPL reference-of-last-resort — Linux i915 display:**
|
||||
- `intel_gmbus.c`, `intel_dp_aux.c` — the concrete EDID/DDC and DP-AUX register sequences.
|
||||
- `intel_ddi.c` / `intel_ddi_buf_trans.c`, `intel_cdclk.c`, `intel_dpll_mgr.c` — DDI/CDCLK/PLL;
|
||||
`i9xx_plane.c`, `intel_crtc.c` — plane/pipe; `intel_dp.c` — link training. Huge and modular; a
|
||||
reference to confirm undocumented quirks, not a template.
|
||||
|
||||
**Permissive prior art — Haiku `intel_extreme` (MIT):**
|
||||
- [`src/add-ons/kernel/drivers/graphics/intel_extreme/`](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)
|
||||
— a second independent modeset-only driver; MIT, so structurally readable for a permissive danos.
|
||||
- [generations.html](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html)
|
||||
— the best plain-English per-generation fault-line map.
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
- **Does the target machine have a usable, cable-attached iGPU at all?** F-SKU check, CPU generation,
|
||||
and monitor cabling must be resolved before any effort estimate is trusted (see
|
||||
[practical caveat](#the-practical-desktop-caveat)).
|
||||
- **Does danos even need native mode-*setting*, or only plane/scanout control on the GOP-set mode?**
|
||||
If runtime mode changes aren't required, the driver collapses to EDID + plane page-flips, dropping
|
||||
the DPLL/DDI/link-training ~70% of the work.
|
||||
- **GGTT vs raw physical:** confirm from the exact target-gen PRM that `PLANE_SURF` is interpreted as
|
||||
a GGTT graphics address (well-established, but per-gen confirmation advisable), and the PTE size /
|
||||
`GTTMMADR` / aperture layout for writing GGTT entries.
|
||||
- **Reuse the firmware/GOP GGTT + framebuffer, or install your own GGTT entries?** The latter (needed
|
||||
for double-buffering) means writing GGTT PTEs from the userspace driver via an MMIO grant.
|
||||
- **eDP panel power sequencing** (`PP_*`, T1–T12 delays) — not covered in this pass and a common
|
||||
black-screen source.
|
||||
- **IOMMU interaction** — whether the display's GGTT-mediated DMA needs IOMMU passthrough for the
|
||||
framebuffer pages under danos's M16 IOMMU, or sits before the IOMMU on the target platform.
|
||||
- **DP link-training / AUX robustness and per-generation register drift** are the dominant *risks* —
|
||||
not documentation scarcity.
|
||||
- **Exact Haswell/BDW MMIO offsets** (commonly cited: GMBUS ~`0xC5100`, `DDI_AUX_CTL_A` ~`0x64010`,
|
||||
`DDI_BUF_CTL_A` ~`0x64000`, `DP_TP_CTL_A` ~`0x64040`) were not extracted verbatim from the PRM —
|
||||
confirm against `i915_reg.h` before coding.
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot; verify against current libgfxinit / i915 source and the specific target
|
||||
generation's PRM before building. Intel's public-PRM coverage and the muxing/F-SKU realities of a
|
||||
given machine both change what is actually achievable.*
|
||||
+20
-9
@@ -9,7 +9,7 @@ reboot is miserable.
|
||||
|
||||
This is the machinery that catches those faults and prints what happened instead.
|
||||
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
||||
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
||||
|
||||
## First the GDT
|
||||
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
|
||||
firmware left a GDT in place, but we don't control it, so we install our own with
|
||||
known selectors: `0x08` kernel code, `0x10` kernel data.
|
||||
|
||||
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||
`system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||
plus code and data — where the only bits that matter in long mode are the access
|
||||
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
||||
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
||||
@@ -33,7 +33,7 @@ into CS:RIP.
|
||||
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
||||
entry is a 16-byte *gate* holding the handler's address (split across three
|
||||
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
||||
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the
|
||||
means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
|
||||
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
||||
(`idt_flush`).
|
||||
|
||||
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
|
||||
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
||||
the #DF handler then tried to push onto that same bad stack, it would fault a
|
||||
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
||||
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid.
|
||||
**IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
|
||||
|
||||
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
||||
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
||||
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
|
||||
|
||||
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
||||
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
||||
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||
in `system/kernel/architecture/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
||||
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
||||
Zig handler with a pointer to the whole thing.
|
||||
@@ -80,11 +80,22 @@ inline). `build.zig` adds `isr.s` to the arch module.
|
||||
## Reporting a fault
|
||||
|
||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||
hook. The generic kernel installs a reporter (`onException` in `main.zig`) that
|
||||
prints, in red, the exception name and vector, the error code, the faulting RIP
|
||||
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||
prints the exception name and vector, the error code, the faulting RIP
|
||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||
**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is
|
||||
terminal; the point is that it's now **visible** instead of a silent reset.
|
||||
**CR2**. What happens next depends on where the fault came from:
|
||||
|
||||
- **User mode (CPL 3): kill the process, keep the machine.** The kernel is intact
|
||||
(the CPU trapped onto the task's kernel stack), so the faulting process is
|
||||
killed — address space, IRQ bindings, and IPC handles reclaimed; a client it
|
||||
owed a reply to is failed with `-EPEER` — and the core reschedules. A crashing
|
||||
driver takes itself down, never the OS. This is fault recovery step 2 of
|
||||
[resilience.md](resilience.md). NMI, double fault, and machine check are
|
||||
excluded: they report machine trouble regardless of what was running.
|
||||
- **Kernel mode: halt this core.** The trusted base itself is broken, so there is
|
||||
nothing safe to kill; the fault is still *contained* to the core (an
|
||||
application-processor fault leaves the rest of the system running), and the
|
||||
report makes it **visible** instead of a silent reset.
|
||||
|
||||
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
||||
caught.
|
||||
|
||||
+79
-12
@@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
|
||||
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
There are two layers, built a milestone apart:
|
||||
|
||||
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||
described below. The primitive, and where the blocking discipline was worked out.
|
||||
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||
second half of this document.
|
||||
|
||||
## The channel
|
||||
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
|
||||
@@ -42,16 +50,75 @@ full and empty over and over, so both the blocking-send and blocking-recv paths
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||
|
||||
## Endpoints: call/reply across address spaces
|
||||
|
||||
A channel connects two kernel threads sharing one address space. Real servers are
|
||||
*processes*, so the payload has to cross an address-space boundary. That's
|
||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||
bounce buffer).
|
||||
|
||||
Two syscalls carry it:
|
||||
|
||||
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||
any), then block for the next request. One syscall, because a server's steady state
|
||||
is *always* "finish the last one, wait for the next".
|
||||
|
||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||
client calls `ipc_lookup(service_id)`.
|
||||
|
||||
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||
the message: the caller's task id.
|
||||
|
||||
### Interrupts are messages too
|
||||
|
||||
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||
client wants something" from "the hardware wants something". Notifications sit in a
|
||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||
elsewhere is not lost.
|
||||
|
||||
This is what makes a user-space driver possible at all, and it's the subject of
|
||||
[drivers.md](drivers.md).
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
- **Across address spaces.** Today both endpoints are kernel threads sharing the
|
||||
kernel's memory, so the message is copied within one address space. When user
|
||||
mode arrives, the same channel carries messages between *isolated* processes,
|
||||
copying the payload across the boundary — which is where IPC earns its place as
|
||||
the microkernel's backbone.
|
||||
- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply)
|
||||
on top of channels, the shape most driver/service calls take.
|
||||
- **Interrupts as messages.** A hardware interrupt delivered to the driver task
|
||||
that owns the device, as an IPC message.
|
||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||
low-priority server doesn't suffer unbounded priority inversion.
|
||||
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||
every capability is either well-known (the registry) or inherited — there's no way
|
||||
to delegate one.
|
||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||
lock; a bulk transfer wants shared pages, not a copy.
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||
notification mechanism:
|
||||
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol message.
|
||||
|
||||
+3
-3
@@ -13,7 +13,7 @@ kernel follows.
|
||||
|
||||
## The log is multi-sink
|
||||
|
||||
`src/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||
registered **sinks**, each best-effort and self-guarding:
|
||||
|
||||
```zig
|
||||
@@ -36,7 +36,7 @@ Properties that matter:
|
||||
## The framebuffer is *not* a log sink
|
||||
|
||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||
Today `src/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||
@@ -58,7 +58,7 @@ screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||
`Framebuffer.present()` (in `src/root.zig`) gates every on-screen path. A headless,
|
||||
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||
serial-less machine boots and runs correctly — it just goes quiet.
|
||||
|
||||
## Last-resort channels (no text output at all)
|
||||
|
||||
+3
-3
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
|
||||
|
||||
## The neutral format
|
||||
|
||||
Defined in `src/root.zig`, the shared loader↔kernel contract:
|
||||
Defined in `system/boot-handoff.zig`, the shared loader↔kernel contract:
|
||||
|
||||
```zig
|
||||
pub const MemoryKind = enum(u32) {
|
||||
@@ -68,7 +68,7 @@ pub const BootInfo = extern struct {
|
||||
|
||||
## The loader side (UEFI)
|
||||
|
||||
Two functions in `src/boot/efi.zig`, called from `exitBootServices`:
|
||||
Two functions in `boot/efi.zig`, called from `exitBootServices`:
|
||||
|
||||
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
||||
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
||||
@@ -121,7 +121,7 @@ The kernel receives a plain array and reads it with zero UEFI knowledge:
|
||||
|
||||
```zig
|
||||
const mm = boot_info.memory_map;
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable_pages += r.pages;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
# Native NVIDIA GPU support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *native* display driver for a
|
||||
real discrete NVIDIA GPU — specifically an **RTX 3060 (Ampere GA106)** — would take, and how it
|
||||
would slot into danos's pluggable scanout architecture. It is a survey of primary sources
|
||||
(NVIDIA's [open-gpu-kernel-modules](https://github.com/NVIDIA/open-gpu-kernel-modules), the Linux
|
||||
[nouveau/nvkm](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/nouveau) driver,
|
||||
NVIDIA's [open-gpu-doc](https://nvidia.github.io/open-gpu-doc/), and
|
||||
[linux-firmware](https://github.com/NVIDIA/linux-firmware)), not an implementation. The NVIDIA
|
||||
driver landscape moves quickly (GSP defaults, firmware ABIs); treat specifics as a mid-decade
|
||||
snapshot and re-verify against current source before building.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the
|
||||
v2 model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- A **minimal display-only driver** (EDID + mode-set + framebuffer scanout, **no** 3D/compute)
|
||||
for the RTX 3060 **can and should avoid the GSP entirely**. nouveau has a register-level,
|
||||
CPU-driven display path for Ampere (`nvkm/engine/disp/ga102.c`) that lights up GA106 with no
|
||||
external firmware; the signed-firmware wall gates the **compute/graphics** engines (PGRAPH),
|
||||
**not** the display controller. "GSP is mandatory on Ampere" is true only for NVIDIA's own
|
||||
RM-object route.
|
||||
- **danos's UEFI GOP boot is the single biggest thing in its favour.** The VBIOS/GOP has already
|
||||
run devinit and brought up the display PLLs, so a driver attaches to a **live, initialized**
|
||||
GA106 — no firmware load, no cold-boot POST, no devinit interpreter. You reprogram a running
|
||||
display rather than bring one up from cold.
|
||||
- It is still a **hard, multi-week-to-months expert effort** (effort tier ≈ 4/5) dominated by
|
||||
NVDisplay channel-DMA programming, SOR/head routing, DisplayPort AUX + link training, and the
|
||||
display supervisor handshake. The GSP/RM route is tier 5 (near-infeasible solo).
|
||||
- The **licensing tension is counterintuitive**: the permissively-licensed reference (NVIDIA
|
||||
open-gpu-kernel-modules, MIT/GPLv2) is the **hard GSP path**; the register-level display code
|
||||
you actually want lives in **GPL nouveau**. See [Licensing](#licensing).
|
||||
- The **window is closing**: GA10x (Ampere) is the *last* NVIDIA family with a register-level
|
||||
display path — Ada (RTX 40) deleted its non-GSP display HAL. Targeting Ampere specifically
|
||||
matters.
|
||||
- **Recommendation:** for *this card*, GOP already gives native-resolution scanout with zero GPU
|
||||
code and zero maintenance. A native driver buys only runtime mode changes, hardware
|
||||
vsync/vblank, and multihead. It is justified if that runtime control is a danos goal, or to
|
||||
*learn the craft* — for which an Intel iGPU or a pre-Turing NVIDIA card reaches "first pixel"
|
||||
far faster.
|
||||
|
||||
## The GSP wall, and why display sits on the near side of it
|
||||
|
||||
On Turing and later, NVIDIA split its driver's Resource Manager into a host **CPU-RM** and a
|
||||
**GSP-RM** running on an on-die RISC-V core ("Peregrine"), talking over RPC
|
||||
([LWN 953144](https://lwn.net/Articles/953144/)). The GSP is a *full resource manager*, not a
|
||||
display coprocessor — there is no "display-only" GSP image and no small display RPC subset. Its
|
||||
boot chain is entirely signed and mandatory: a VBIOS-resident **FWSEC-FRTS** app carves a
|
||||
write-protected region (WPR2), a signed **Booter** on the SEC2 falcon loads the GSP bootloader,
|
||||
and that loads **GSP-RM** inside WPR. The firmware ships pre-computed signatures and the driver
|
||||
picks one by an on-chip fuse-version register — **you cannot self-sign**, and there is **no stable
|
||||
firmware ABI** (it is revised every driver release; nouveau and the Rust nova-core driver each pin
|
||||
exactly one version). A GSP driver is a permanent maintenance liability, not a one-time build
|
||||
([LWN 1037379](https://lwn.net/Articles/1037379/),
|
||||
[nova-core cover letter](https://lore.freedesktop.org/nouveau/20250826-nova_firmware-v2-7-93566252fe3a@nvidia.com/T/)).
|
||||
|
||||
**But display doesn't need any of that on Ampere.** `nvkm/engine/disp/ga102.c` dual-dispatches:
|
||||
|
||||
```
|
||||
if (nvkm_gsp_rm(device->gsp)) return r535_disp_new(&ga102_disp, ...); // GSP RPC path
|
||||
return nvkm_disp_new_(&ga102_disp, ...); // direct register path
|
||||
```
|
||||
|
||||
Both branches use the same `ga102_disp` HAL and the same `GA102_DISP_*` class IDs; GSP merely
|
||||
swaps register programming for RPC. GA106 (chipset `0x176`) is wired to `ga102_disp_new` in the
|
||||
device table, identical to GA102/103/104/107. Ampere lit up displays via the **direct** path in
|
||||
Linux 5.11/5.17 — two years before GSP-RM landed (6.7, 2023)
|
||||
([ga102.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/ga102.c),
|
||||
[Phoronix GA106](https://www.phoronix.com/news/Nouveau-NVIDIA-GA106)).
|
||||
|
||||
**Caveat — this is now the legacy path.** As of Linux 6.18, nouveau defaults to GSP on
|
||||
Turing/Ampere; the direct path is a retained, forceable fallback (`nouveau.config=NvGspRm=0`, and
|
||||
automatic when GSP firmware is absent). It is stable and proven, but NVIDIA and nova-core are
|
||||
moving to GSP-only, and **Ada already deleted its non-GSP display HAL**. GA10x is the last family
|
||||
that keeps a register-level display path.
|
||||
|
||||
## What "direct" actually entails
|
||||
|
||||
"Direct" is not "plain register pokes." Only SOR / PLL / DP-link / clock setup is bare MMIO. The
|
||||
**mode-set and scanout themselves flow through the NVDisplay channels — a DMA pushbuffer**:
|
||||
|
||||
- Display classes for Ampere (the C670 family): core `GA102_DISP_CORE_CHANNEL_DMA` (`0xc67d`),
|
||||
window `0xc67e`, window-immediate `0xc67b`, cursor `0xc67a` (headers `clc67d.h` / `clc67e.h` /
|
||||
`clc67a.h` in [open-gpu-doc `classes/display/`](https://github.com/NVIDIA/open-gpu-doc/tree/master/classes/display)).
|
||||
- The core channel needs **instance memory, a RAMHT, DMA objects, and a channel user-MMIO
|
||||
region** ([disp/chan.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/chan.c)).
|
||||
The register-level "plumbing" to allocate/kick a channel is in NVIDIA's GA102 display register
|
||||
manual: `NV_PDISP_FE_CHNCTL_CORE/WIN/CURS`, `NV_PDISP_FE_PBBASE/PBBASEHI`
|
||||
([dev_display_withoffset.ref.txt](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)).
|
||||
- **Mode-set is a method stream** on the core channel: `HEAD_SET_RASTER_*`,
|
||||
`HEAD_SET_PIXEL_CLOCK_FREQUENCY`, `HEAD_SET_CONTROL_OUTPUT_RESOURCE`, `SOR_SET_CONTROL`
|
||||
(protocol select), viewport/scaler, then `UPDATE`. The window channel points at the scanout
|
||||
surface (`SET_CONTEXT_DMA_ISO`, `SET_STORAGE`, `SET_OFFSET`).
|
||||
- After `UPDATE` you must complete the display **supervisor** interrupt handshake (SV1/SV2/SV3).
|
||||
|
||||
**EDID and DisplayPort are a separate subdev you must port.** open-gpu-doc documents *none* of
|
||||
EDID/DDC/AUX. On the direct path you read EDID in-driver via nouveau's `nvkm/subdev/i2c`: bit-bang
|
||||
**DDC/I²C at address `0x50`** (E-DDC `0x30`) for TMDS/HDMI, or native **DP AUX** in `i2c/aux.c`
|
||||
for DisplayPort. DisplayPort **link training** (the `dp.c` `train_cr` / `train_eq` state machine
|
||||
over AUX — clock recovery, lane/rate, voltage-swing/pre-emphasis) is the single hardest and most
|
||||
fragile piece; a DVI/HDMI (TMDS) panel avoids it entirely.
|
||||
|
||||
## The memory floor (smaller than you'd fear)
|
||||
|
||||
Neither route hands you a framebuffer allocator — even GSP-RM does not manage the scanout
|
||||
framebuffer; the driver owns VRAM and merely tells GSP where its page directory is. But
|
||||
display-only is a small fraction of a full GEM/TTM stack:
|
||||
|
||||
- **Pitch-linear (untiled) scanout is allowed** on nv50→Ampere — the window's storage method has a
|
||||
`PITCH` layout mode, so you skip block-linear tiling math
|
||||
([wndwc37e.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/dispnv50/wndwc37e.c)).
|
||||
- The window references its surface through a simple **display context-DMA**
|
||||
(`SET_CONTEXT_DMA_ISO` + a 256-byte-granular `SET_OFFSET = addr>>8`) — a base/limit descriptor,
|
||||
**not** the GPU's 5-level compute page tables. **No full GPU VMM is needed** for scanout.
|
||||
- The surface must live in **VRAM** in practice (nouveau always pins scanout to VRAM). *Open
|
||||
question:* whether GA10x can scan out from a system-memory (GART) surface via a sysmem-target
|
||||
ctxdma — which would let danos skip a VRAM allocator. No source forbids it; nouveau never does
|
||||
it (confidence: medium).
|
||||
- **CPU access** to the framebuffer for compositing goes through **BAR1** (a VRAM aperture); BAR0
|
||||
is the 16 MB register window. BAR1 can be smaller than 12 GB of VRAM unless Resizable BAR maps
|
||||
it all.
|
||||
|
||||
**Net:** you need (1) a contiguous aligned VRAM allocator (256-byte base, pitch a multiple of
|
||||
64 bytes — confirm against the Ampere display refs), (2) a little instmem for the channel
|
||||
pushbuffers + iso ctxdma, (3) a BAR1 CPU mapping. You do **not** need the 5-level VMM, GEM/TTM
|
||||
eviction, or tiling.
|
||||
|
||||
## Licensing
|
||||
|
||||
The tension is the opposite of convenient:
|
||||
|
||||
- **NVIDIA open-gpu-kernel-modules is dual MIT/GPLv2** — usable under MIT, no copyleft on your
|
||||
other code — **but its display logic is the GSP/RM-object route.** Its class headers
|
||||
(`cl0073.h`, `cl2080.h`, `ctrl0073*.h`) are useful, permissive references.
|
||||
- **nouveau is GPLv2**, and the **register-level display sequences you actually want live in
|
||||
nouveau**, not in the MIT code. So the *easy technical path is the GPL-licensed one.* Reading
|
||||
GPL nouveau and reimplementing it in Zig is a derivative-work risk proportional to how closely
|
||||
your code tracks its structure/constants.
|
||||
|
||||
Options: **(a)** accept that the danos NVIDIA display driver is a **GPL component**. danos's
|
||||
userspace-driver-over-IPC model (a driver is a separate process behind a defined protocol, not
|
||||
linked into the kernel) is about the cleanest possible GPL boundary, so the GPL would be contained
|
||||
to that one binary and the rest of danos could keep its own license — but this is a
|
||||
licensing-boundary judgement that wants real diligence, not a settled fact. **(b)** clean-room
|
||||
from *specification* rather than *code*: [envytools](https://envytools.readthedocs.io) + NVIDIA's
|
||||
open-gpu-doc register manuals + the MIT OGKM class headers, treating nouveau as
|
||||
documentation-of-last-resort.
|
||||
|
||||
**Firmware licensing is moot for the direct path** (no firmware is loaded). For completeness: the
|
||||
GSP blobs are marked redistributable under `LICENCE.nvidia`, which permits use by **any
|
||||
OSI-approved open-source OS** (not just Linux), on NVIDIA GPUs, **unmodified**, with **no
|
||||
reverse-engineering of the firmware binary**. The one gate — is danos released under an OSI
|
||||
license? — is only reached on the GSP route, which this doc recommends against for this card.
|
||||
|
||||
## Prior art
|
||||
|
||||
**No one has built a from-scratch native NVIDIA driver outside Linux.** FreeBSD ships
|
||||
`nvidia-drm-kmod`, a *port of NVIDIA's own closed `nvidia-drm.ko`* loading the GSP blob (its old
|
||||
nouveau port was removed). Haiku's NVIDIA support is likewise a *port of OGKM* (GSP, Turing+, very
|
||||
alpha). OpenBSD / DragonFly have neither. Every non-Linux OS that supports modern NVIDIA chose to
|
||||
**wrap NVIDIA's GSP stack** rather than write a native driver. A danos direct-register driver
|
||||
would have exactly one reference implementation — GPL nouveau — and no non-Linux precedent.
|
||||
|
||||
## Alternatives
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; **no runtime mode change, no hardware vsync, no multihead** |
|
||||
| **Pre-Turing NVIDIA** (Kepler / early Maxwell) | Direct EVO/disp-core + CRTC/PLL modeset, **no signed firmware, no coprocessor**; mature nouveau reference | Older display class; not this card; only reclocking is firmware-gated |
|
||||
| **Intel iGPU** | **Publicly documented** register interfaces (Intel PRMs); no coprocessor mediating modeset | i915 is huge + generation-specific; write one generation from the PRM |
|
||||
| **Native GA106 direct** (this doc) | Runtime modeset, vsync, multihead on the actual card | Tier-4 effort; GPL reference; DP link training; legacy/de-emphasized path |
|
||||
| **GA106 via GSP/OGKM** | Also unlocks 3D / reclocking later | Tier-5; ~14k-line ante; unstable version-pinned ABI; unprecedented outside Linux |
|
||||
|
||||
## "First light" milestones (direct path, inheriting GOP state)
|
||||
|
||||
Framed as a danos `.scanout` service (like the virtio-gpu driver), taking the direct register path
|
||||
and inheriting the GOP-initialized display — no signed firmware, no devinit, no GSP:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate GA106 (`0x176`), map **BAR0** (registers) and **BAR1** (VRAM
|
||||
aperture) via danos MMIO grants; confirm the display engine is GOP-live.
|
||||
2. **VRAM + instmem allocator** — contiguous aligned VRAM for the scanout surface (256-byte base)
|
||||
+ small instmem for pushbuffers / RAMHT / iso ctxdma. No VMM, no TTM.
|
||||
3. **EDID** — port `nvkm/subdev/i2c` DDC (`0x50`) + DP-AUX (`aux.c`); read + parse the panel EDID.
|
||||
4. **Core channel up** — allocate the `0xc67d` core channel as a DMA pushbuffer; stand up the
|
||||
SV1/SV2/SV3 supervisor-interrupt handshake.
|
||||
5. **First pixel = reprogram, don't re-POST** — bind a window (`0xc67e`) at the existing WC
|
||||
framebuffer via `SET_CONTEXT_DMA_ISO` + `SET_OFFSET`, pitch-linear, `UPDATE`; prove you can
|
||||
drive the *current* GOP mode from your own channel before changing anything.
|
||||
6. **Modeset** — push raster timings on a head, route head→SOR→connector, program the pixel-clock
|
||||
PLL, switch to an EDID mode (needs the `clc67d/e` method opcodes from the OGKM headers + the
|
||||
supervisor timing from nouveau `head.c`).
|
||||
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||
`dp.c` `train_cr`/`train_eq` state machine. TMDS/HDMI is far simpler.
|
||||
8. **Wire into the compositor `.scanout` backend** (`attach_scanout`), add vsync via the display
|
||||
interrupt, then multihead.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||
working display (exactly the resilience v2 already provides via re-attach).
|
||||
|
||||
## Reading list
|
||||
|
||||
**Direct path — nouveau (GPLv2):**
|
||||
- `nvkm/engine/disp/ga102.c` — the GA10x display HAL + the GSP/non-GSP dispatch.
|
||||
- `nvkm/engine/disp/{head.c, ior.c, dp.c, hdmi.c, chan.c}` — head/SOR routing, DP AUX + link
|
||||
training, channel-DMA plumbing.
|
||||
- `dispnv50/{corec37d.c, corec57d.c, wndwc37e.c, wndwc57e.c, wndwc67e.c, headc37d.c, cursc37a.c}`.
|
||||
- `nvkm/subdev/i2c` (DDC + `aux.c`) for EDID; `nvkm/subdev/bios/init.c` + `devinit/` **only** if
|
||||
you ever have to re-POST (danos's GOP handoff means you shouldn't).
|
||||
|
||||
**Object model / GSP path — NVIDIA OGKM (MIT/GPLv2):** class headers `cl0073.h`, `cl2080.h`,
|
||||
`ctrl0073system.h`, `ctrl0073specific.h`; `src/nvidia/` for RM control sequences.
|
||||
`nvidia-modeset.ko` (NVKMS) is a *policy* layer over RM and can be bypassed entirely.
|
||||
[nova-core](https://lore.freedesktop.org/nouveau/) (Rust) is the forward-looking reference for GSP
|
||||
boot mechanics (falcon signing, queue rings, RPC).
|
||||
|
||||
**Register / method specs — NVIDIA open-gpu-doc:**
|
||||
- [`classes/display/README.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/README.txt)
|
||||
— the channel model + class-to-GPU map (read first).
|
||||
- [`classes/display/clc67d.h`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/clc67d.h)
|
||||
+ `clc67e.h` / `clc67a.h` — the Ampere core/window/cursor mode-set method vocabulary.
|
||||
- [`manuals/ampere/ga102/dev_display_withoffset.ref.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)
|
||||
— `NV_PDISP_FE_*` channel/pushbuffer registers + SOR.
|
||||
- [`DCB`](https://github.com/NVIDIA/open-gpu-doc/tree/master/DCB) — connector→output-resource
|
||||
routing; [`Devinit`](https://github.com/NVIDIA/open-gpu-doc/tree/master/Devinit) +
|
||||
[`BIOS-Information-Table`](https://github.com/NVIDIA/open-gpu-doc/tree/master/BIOS-Information-Table)
|
||||
— VBIOS parsing (bring-up reference; not needed if inheriting GOP).
|
||||
- The 632 KB Volta [`dev_display.ref`](https://download.nvidia.com/open-gpu-doc/Display-Ref-Manuals/1/gv100/dev_display.ref)
|
||||
is the best shot at SOR-DP/AUX register detail the smaller Ampere file omits.
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
Each needs a direct read of the named nouveau file or experimentation on the actual card:
|
||||
|
||||
- Exact GA106 register/method offsets and PADLINK→SOR→connector wiring (can vary by board vendor).
|
||||
- Whether *any* PLL/devinit re-run is unavoidable vs. fully inherited from GOP.
|
||||
- Whether DisplayPort needs full retraining on takeover, or the GOP-established link can be reused.
|
||||
- The precise SV1/SV2/SV3 supervisor sequence.
|
||||
- Whether a system-memory-target scanout ctxdma could eliminate the VRAM allocator.
|
||||
- The exact `clc67d.h`/`clc67e.h` method opcode numbers (not captured verbatim in the survey).
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot; verify against current nouveau / open-gpu-kernel-modules source before
|
||||
building — NVIDIA's GSP defaults and firmware ABIs change per release.*
|
||||
@@ -0,0 +1,9 @@
|
||||
# OS Developer Guide
|
||||
|
||||
This document is for those who need to understand the architectural decisions behind the OS.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The os was initially written in zig because it has excellent support for EFI. With zig, we could forgo using a third party bootloader, reducing the time to boot up the kernel. Following the "Zen of Zig", helped to produce the most readable codebase for an operating system ever created. So those, new to OS development could quickly get up to speed.
|
||||
|
||||
|
||||
+49
-14
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
|
||||
them, and — crucially — maps with **real permissions**.
|
||||
|
||||
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
||||
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`.
|
||||
behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||
|
||||
## The format
|
||||
|
||||
@@ -17,20 +17,54 @@ the final 4 KiB page. Each entry holds a physical address plus flag bits —
|
||||
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
||||
pages: precise, and the extra table memory is negligible against available RAM.
|
||||
|
||||
## Higher half: the address-space layout
|
||||
|
||||
danos is a **higher-half kernel**. The kernel is linked to run at
|
||||
`0xFFFF_FFFF_8000_0000` but loaded low (the linker script's `AT()` gives each
|
||||
segment a physical load address at 1 MiB up; the bootloader maps the high link
|
||||
address to the low load address in its bootstrap tables and jumps in). The entire
|
||||
**low canonical half is reserved for user space**; the kernel lives in the top half
|
||||
alongside a **physmap** — a straight window onto all of physical memory at
|
||||
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||
|
||||
| region | virtual base | PML4 slot |
|
||||
|--------|--------------|-----------|
|
||||
| user image + stack | `0x0000_7000_0000_0000` | 224 (low half) |
|
||||
| kernel heap | `0xFFFF_8000_0000_0000` | 256 |
|
||||
| physmap (all RAM + MMIO windows) | `0xFFFF_8800_0000_0000` + phys | 272 |
|
||||
| kernel image | `0xFFFF_FFFF_8000_0000` | 511 |
|
||||
|
||||
The bootloader builds temporary **bootstrap tables** (identity + a 4 GiB physmap +
|
||||
the high kernel) so it can switch CR3 and jump to the high entry; the kernel then
|
||||
builds its own precise tables below and abandons them. Because both use the same
|
||||
`physmap_base`, any physmap pointer minted before the switch stays valid after it.
|
||||
|
||||
## What gets mapped, and with what permissions
|
||||
|
||||
The address space is built in three passes (`init`):
|
||||
The address space is built in four passes (`init`):
|
||||
|
||||
1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the
|
||||
[memory map](memory-map.md) is mapped virtual == physical, read-write and
|
||||
*non-executable*. Identity mapping keeps everything already running valid across
|
||||
the CR3 switch (the frame allocator addresses frames by physical address, page
|
||||
tables are reached the same way, the stack stays put).
|
||||
2. **The framebuffer and the Local APIC**, the device memory we actually touch,
|
||||
also RW + NX. Everything else — unbacked address space, other MMIO — is simply
|
||||
left unmapped, so a stray access faults instead of silently succeeding.
|
||||
3. **The kernel's own segments, overlaid with their true ELF permissions.** This is
|
||||
1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the
|
||||
[memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and
|
||||
*non-executable*. There is **no low/identity mapping** — the low half is user
|
||||
space. (Frames the kernel touches while still building these tables are reached
|
||||
through the loader's bootstrap physmap, which covers the low 4 GiB; both the
|
||||
frame allocator and the table builder scan low-address-up, so those frames stay
|
||||
under that limit.)
|
||||
2. **The framebuffer and the Local APIC**, the device memory the kernel touches
|
||||
directly, as physmap windows (RW + NX). Other MMIO is mapped on demand by
|
||||
`mapMmio`, also into the physmap; everything else is left unmapped, so a stray
|
||||
access faults instead of silently succeeding.
|
||||
3. **The kernel's own segments, overlaid with their true ELF permissions**, at
|
||||
their high link addresses mapped to their low physical load addresses. This is
|
||||
the interesting part.
|
||||
4. **Every higher-half PML4 entry pre-created** (an empty PDPT where none exists
|
||||
yet). The kernel half is then a fixed set of top-level slots, so a per-process
|
||||
address space can share it by copying `PML4[256..512)` once — growth beneath
|
||||
those slots (heap, on-demand MMIO) propagates to every address space because
|
||||
they share the PDPTs. `init` asserts no new higher-half PML4 entry appears
|
||||
afterward.
|
||||
|
||||
### W^X from the ELF program headers
|
||||
|
||||
@@ -55,9 +89,10 @@ reserved bit and fault.
|
||||
|
||||
### The null guard
|
||||
|
||||
Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now
|
||||
takes a page fault instead of quietly reading or writing real memory — turning a
|
||||
whole class of silent bugs into an immediate, located crash.
|
||||
The whole low half is unmapped except for explicit user mappings, so page 0 (and
|
||||
every near-null address) is unmapped by construction. A null (or near-null) pointer
|
||||
dereference in the kernel takes a page fault instead of quietly reading or writing
|
||||
real memory — turning a whole class of silent bugs into an immediate, located crash.
|
||||
|
||||
## Switching on, and the on-demand API
|
||||
|
||||
|
||||
+128
@@ -0,0 +1,128 @@
|
||||
# The power service: events and shutdown
|
||||
|
||||
A laptop lid closes, a battery drains, someone presses the power button — and
|
||||
several parts of the system might care: a session manager dims the screen, a
|
||||
logger notes it, and ultimately *something* has to turn the machine off. None of
|
||||
them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
Where the events come from is firmware-specific — on x86 they ride the ACPI SCI
|
||||
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
This is why the protocol is `power`, not "ACPI events": naming a cross-firmware
|
||||
surface after one firmware would leak x86 into code the ARM port must reuse
|
||||
unchanged.
|
||||
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
|
||||
| Direction | Operation | Purpose |
|
||||
|---|---|---|
|
||||
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||
|
||||
Events are published, not polled: like the input service, the service holds
|
||||
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||
message, so a slow or dead subscriber can never wedge the source. The event
|
||||
vocabulary is hardware-neutral:
|
||||
|
||||
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||
- `notify` — a device notification that maps to none of the above; its `code`
|
||||
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||
and what happened.
|
||||
|
||||
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||
generic `notify` is fully described without a second round trip.
|
||||
|
||||
**`shutdown` is authority, not information.** It is the only operation that
|
||||
*does* something irreversible, so it is gated: the contract is that only init
|
||||
(PID 1) may request it, because init is the process that has already run the stop
|
||||
sequence over everything else. The acpi service implements this as a **soft
|
||||
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||
init is the one subscriber. That stands in for "only the system supervisor may
|
||||
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||
is not init.
|
||||
|
||||
## Orderly shutdown
|
||||
|
||||
Powering off cleanly is where the power service, the [process
|
||||
lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init
|
||||
already supervises the services it starts; for shutdown it runs **one event loop
|
||||
over one endpoint** that carries three things at once: its children's exit
|
||||
notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||
**power events** it subscribes to — plus a re-arming heartbeat timer proving PID
|
||||
1 is alive. (init subscribes with retries, because the power service registers
|
||||
`.power` well after init starts; a missing power service is not fatal — a
|
||||
`terminate` signal drives the same path.)
|
||||
|
||||
On a `power_button` event or a `terminate` signal, init:
|
||||
|
||||
1. logs that it is shutting down,
|
||||
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||
last (other services may flush through it), each child getting the
|
||||
*terminate → deadline → kill* escalation from
|
||||
[process-lifecycle.md](process-lifecycle.md), and
|
||||
3. requests `.power` `shutdown`.
|
||||
|
||||
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||
the machine off, it logs loudly so a test fails rather than hangs.
|
||||
|
||||
**No new system call was needed for S5.** The broad io_port grant on the
|
||||
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||
could physically already do; formalizing it as a protocol operation added a
|
||||
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||
panic-time poweroff, where no user space is available to ask.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Two QEMU scenarios exercise the path, both injecting a real ACPI power-button
|
||||
press via QMP `system_powerdown` (there is no other deterministic power event on
|
||||
this config):
|
||||
|
||||
- `power-button` proves the source: the acpi service's SCI handler logs the
|
||||
press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)).
|
||||
- `orderly-shutdown` proves the whole composition: button → init logs shutting
|
||||
down → children stopped → the service enters S5 → QEMU exits. The ordered
|
||||
regex is the proof, and QEMU's self-exit through S5 is the pass.
|
||||
|
||||
## Scope
|
||||
|
||||
Interface-complete but validated on real hardware (the author's laptop) later,
|
||||
because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the
|
||||
interface stubs, lid and AC events, and the embedded controller's `_Qxx`
|
||||
queries. Deliberately out of scope for now: reboot over the power protocol, S3
|
||||
sleep, per-device D-states (a future lifecycle-vocabulary extension, since
|
||||
"suspend" has the shape of a signal every driver must answer and has no consumer
|
||||
until laptop sleep), and thermal zones.
|
||||
|
||||
## See also
|
||||
|
||||
- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power
|
||||
button fixed event, and GPE/Notify dispatch in the acpi service.
|
||||
- [discovery.md](discovery.md) — why the surface is domain-named, and the
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
@@ -0,0 +1,327 @@
|
||||
# Process lifecycle: signals over IPC
|
||||
|
||||
**Status: increments 1–4 built** (2026-07-12): claim release on death, exit
|
||||
reasons, published exit events, and signals + one-shot timers + the service
|
||||
harness are all in — the interface below is as-built. The primitives underneath
|
||||
predate this design ([process-management.md](process-management.md):
|
||||
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||
reload, and die the same way. The device manager is simply this design's first
|
||||
serious customer ([device-manager.md](device-manager.md)).
|
||||
|
||||
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||
was a mistake) without inheriting the mechanism, the API, or the names. The naming
|
||||
rule is danos's own and it is strict: plain words that communicate intent
|
||||
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||
|
||||
## Why a standard vocabulary
|
||||
|
||||
A supervisor can only manage processes it has never heard of if "please exit" means
|
||||
the same thing to all of them. That is the one thing POSIX signals got deeply right:
|
||||
`SIGTERM` means the same thing to nginx and to a five-line script, which is why
|
||||
process supervision on Unix (init systems, container runtimes) is possible at all.
|
||||
danos wants that property from day one, because supervision-and-restart is the
|
||||
system's core motivation ([resilience.md](resilience.md)).
|
||||
|
||||
What POSIX got wrong — for a system like this — is the **delivery mechanism**:
|
||||
asynchronous control-flow hijack. A Unix handler runs on a stolen stack at an
|
||||
arbitrary instruction boundary, which is why the async-signal-safe function list
|
||||
exists, why `errno` must be saved, and why the canonical signal bug is a SIGTERM
|
||||
handler innocently calling `printf` mid-`malloc`. That entire bug class comes from
|
||||
the mechanism, not the vocabulary, and none of it is worth importing.
|
||||
|
||||
A microkernel already has the right channel: **a signal is a message.** QNX delivers
|
||||
POSIX signals over its message passing; seL4 has notification objects; Erlang turned
|
||||
"death is a message to whoever linked" into a reliability philosophy. danos has
|
||||
already done it once without naming it: a child's death arrives as a notification
|
||||
badge on the supervisor's endpoint — the microkernel's SIGCHLD, the IRQ-as-IPC
|
||||
pattern reused. Signals are the same pattern reused a third time.
|
||||
|
||||
## The mechanism
|
||||
|
||||
- **`signal_bind(endpoint)`** — a process nominates the endpoint its signals arrive
|
||||
on, exactly as `irq_bind` nominates where a device's interrupts land. The runtime
|
||||
does this at startup for any program that opts in.
|
||||
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||
- **Authority**: the supervisor may signal its children — the same link that is
|
||||
already the kill authority. A process may signal itself. Anything broader waits
|
||||
for transferable process handles.
|
||||
- **No binding, no problem**: a process that never calls `signal_bind` is not
|
||||
broken — its signals pend unread and only `process_kill` works on it. Simple
|
||||
programs stay simple; the vocabulary is opt-in, the kill authority is not.
|
||||
|
||||
Because delivery is a message into the process's own event loop, there is no
|
||||
async-signal-safe list in danos: a handler is ordinary code running at a point the
|
||||
process chose. The bug class is gone by construction, not by discipline.
|
||||
|
||||
## The vocabulary: POSIX.1-1990, sorted honestly
|
||||
|
||||
The full 1990 set, and what each becomes. Two intrinsically problematic cases get a
|
||||
defense below the table.
|
||||
|
||||
| POSIX.1-1990 | danos disposition | Notes |
|
||||
|---|---|---|
|
||||
| SIGTERM | signal `terminate` | finish up and exit; the supervisor's polite half |
|
||||
| SIGHUP | signal `reload` | re-read configuration / re-scan |
|
||||
| SIGINT | signal `interrupt` | interactive interrupt; meaningful once a console can send it, in the vocabulary now so numbering is stable |
|
||||
| SIGQUIT | signal `quit` | as SIGINT, without the core-dump baggage |
|
||||
| SIGALRM | signal `alarm` | timer expiry as a message; the Unix SIGALRM+`longjmp` timeout hacks are impossible here. In the vocabulary, unbuilt: no consumer yet, and when one appears it is runtime sugar over the existing timer — zero kernel work |
|
||||
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||
| SIGPIPE | **an error return**, not a signal | see below |
|
||||
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||
|
||||
**The fault signals (SIGSEGV, SIGILL, SIGFPE) are intrinsically wrong for messages.**
|
||||
They are *synchronous* — raised at a specific faulting instruction, not "sometime
|
||||
soon". A message cannot be delivered to a process whose next instruction re-faults;
|
||||
it never reaches its event loop to read it. POSIX only makes fault handlers "work"
|
||||
via the async hijack (run the handler *instead of* the instruction), and even there,
|
||||
returning from a SIGSEGV handler without curing the cause is undefined behavior.
|
||||
danos's architecture already has the better answer: fault → the kernel kills the
|
||||
process ([resilience.md](resilience.md) step 2, built) → the supervisor reads the
|
||||
reason → restart. Recovery is restart, not a handler. This is also truer to the 1990
|
||||
standard than handling is: the standard's default action for all three was
|
||||
"terminate the process".
|
||||
|
||||
**SIGPIPE deserves special contempt.** Its default kills a process that writes to a
|
||||
closed pipe — which is why "the whole server died because one client disconnected"
|
||||
is roughly every network daemon's first production bug, and why every mature codebase
|
||||
contains the same fix: ignore SIGPIPE, handle the `EPIPE` error return. danos made
|
||||
the right choice natively already — a reply owed to a dead peer fails with `-EPEER`.
|
||||
Errors from operations are error returns from those operations. The posix layer can
|
||||
synthesize SIGPIPE for ported code that expects it.
|
||||
|
||||
### Statements, not questions
|
||||
|
||||
A signal and a protocol message both travel over IPC — the difference is the
|
||||
**contract**, not the transport. danos IPC has two primitives, both already in
|
||||
daily use: the **asynchronous notification** (a badge — bits that coalesce into a
|
||||
pending mask; the sender never blocks; no payload, *no reply path*; how IRQs and
|
||||
exit events arrive) and the **synchronous call** (a rendezvous — payload both
|
||||
ways, the caller waits for the reply; how VFS requests work). A signal is the
|
||||
first kind: a *statement*. `terminate` wants no reply — the exit notification is
|
||||
its acknowledgement.
|
||||
|
||||
A health probe is the second kind: a *question*, worthless without its answer —
|
||||
and the answer's absence within a deadline is the very thing being measured.
|
||||
Asked as a signal it has no reply channel (a coalescing bit can't carry an answer,
|
||||
and the authority rule forbids a child signalling its supervisor back); asked as a
|
||||
call, the timeout-is-the-diagnosis semantics come free. So there is no `health`
|
||||
signal. Liveness is the common **`ping`**: a reserved request every harness-run
|
||||
service answers automatically on its main endpoint — still free for the service
|
||||
author, still one obvious way — and a supervisor's probe is a `ping` call with a
|
||||
deadline.
|
||||
|
||||
## The two iron rules
|
||||
|
||||
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||
claims and MSI vectors** (the known gap in
|
||||
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||
work.
|
||||
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||
backoff), killed (the supervisor did it). The exit notification today carries
|
||||
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||
|
||||
## Who learns of a death
|
||||
|
||||
A death has three audiences, and conflating them is how systems end up with either
|
||||
zombie state or privileged snooping:
|
||||
|
||||
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||
audience that needs the *reason*, because it is the only one deciding whether to
|
||||
restart.
|
||||
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||
the same way. This covers the *synchronous* case only.
|
||||
3. **The subscribers** — the new piece, and it is the input service's
|
||||
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: the VFS holds a dead
|
||||
client's open file handles, the input service holds its subscriptions, a future
|
||||
network stack holds its sockets. None of these are the client's supervisor, and
|
||||
none learn anything from a failed reply if the client simply never calls again.
|
||||
So the kernel **publishes every exit** to whoever subscribed:
|
||||
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||
subscriber filters for ids it holds state for and releases what the dead client
|
||||
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||
state by all along is the id the exit event carries.
|
||||
|
||||
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||
non-blocking coalescing notification as everything else — a dying process never
|
||||
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||
|
||||
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||
its clients cleaning up after themselves.** Handle release on client death is the
|
||||
service's job, triggered by the published exit event — never by a courtesy
|
||||
"closing now" message that a crashed client will never send.
|
||||
|
||||
## The stable interface: `runtime.process`
|
||||
|
||||
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||
argv contract). It grows to own the other end of life.
|
||||
|
||||
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||
not make system calls — they call the runtime library, and the system-call numbers,
|
||||
notification bits, and signal bit positions beneath it are a **private kernel ↔
|
||||
runtime contract** that may change at any time (settled 2026-07-12). This is why
|
||||
the runtime exists. Today kernel and runtime ship from one tree in one image, so
|
||||
"stability" is simply building them together. When driver binaries start shipping
|
||||
as separately-versioned applications — the whole point of the restart design — the
|
||||
binary's embedded runtime version becomes compatibility metadata (the same idea as
|
||||
the protocol version in the device manager's `hello`), and the kernel refuses what
|
||||
it cannot serve. Signals therefore need no reserved numbering scheme: the enum
|
||||
below is vocabulary, not ABI.
|
||||
|
||||
```zig
|
||||
/// The signal vocabulary. The value is the bit position in the pending mask — a
|
||||
/// private kernel/runtime detail, free to change while they ship together.
|
||||
pub const Signal = enum(u5) {
|
||||
terminate = 0, // SIGTERM: finish up and exit
|
||||
reload = 1, // SIGHUP: re-read configuration
|
||||
interrupt = 2, // SIGINT
|
||||
quit = 3, // SIGQUIT
|
||||
alarm = 4, // SIGALRM
|
||||
user_1 = 5, // SIGUSR1
|
||||
user_2 = 6, // SIGUSR2
|
||||
};
|
||||
|
||||
/// A decoded pending mask: the coalesced set of signals a notification delivered.
|
||||
pub const SignalSet = struct {
|
||||
pending: u32,
|
||||
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||
};
|
||||
|
||||
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||
/// runtime's service harness calls this; a bare program may call it directly and
|
||||
/// fold signals into its own replyWait loop.
|
||||
pub fn bindSignals(endpoint: usize) bool { ... }
|
||||
|
||||
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||
/// notification (mirrors ipc.Received.isChildExit).
|
||||
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||
|
||||
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||
|
||||
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||
/// notification, then process_kill. The one call a supervisor needs.
|
||||
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||
|
||||
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||
/// process id — the same encoding a supervisor's exit notification uses, decoded
|
||||
/// by the same ipc.Received helpers. For stateful services: release what the dead
|
||||
/// client held (file handles, subscriptions, sockets). Ungated, like
|
||||
/// process_enumerate.
|
||||
pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||
|
||||
/// How a process ended — queried after the exit notification (the kernel records
|
||||
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||
pub const ExitReason = enum(u8) {
|
||||
exited, // returned from main / clean exit
|
||||
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||
segmentation_fault, // SIGSEGV's ghost
|
||||
illegal_instruction, // SIGILL's ghost
|
||||
arithmetic_fault, // SIGFPE's ghost
|
||||
protection_fault, // general protection fault
|
||||
fault, // any other CPU exception
|
||||
killed, // process_kill
|
||||
};
|
||||
```
|
||||
|
||||
Two deliberate absences. There is no `mask`/`block` API — a process that is not
|
||||
ready for a signal simply has not waited on its endpoint yet; the pending mask *is*
|
||||
the blocked set. And there is no per-signal handler registration at this layer —
|
||||
dispatch is the process's own `switch` over `SignalSet`, or the service harness's
|
||||
callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||
|
||||
### The service harness
|
||||
|
||||
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||
harness. A process that bypasses the harness and ignores its signals meets the
|
||||
deadline-then-kill escalation — you cannot force a process to implement an
|
||||
interface, but you can make compliance free and non-compliance fatal.
|
||||
|
||||
### The musl layer later
|
||||
|
||||
The POSIX C layer is a **musl port**: musl's arch/syscall layer retargeted so that
|
||||
what musl believes are kernel syscalls become danos runtime calls and IPC — `open`
|
||||
and `read` onto the VFS protocol, `kill`/`sigaction`/`waitpid` onto this document's
|
||||
vocabulary, `exit` onto the runtime's exit path. `sigaction` handlers registered
|
||||
through it are invoked by the runtime's loop when the signal message arrives —
|
||||
synchronous underneath, async-looking to ported code, delivered at wait boundaries
|
||||
the way most Unix programs already experience signals (at syscalls). No stack hijack
|
||||
ever happens, `SA_RESTART` semantics come free because nothing was interrupted, and
|
||||
SIGPIPE can be synthesized from `-EPEER` for the programs that expect it. C programs
|
||||
get POSIX; danos-native programs never pay for it.
|
||||
|
||||
## Increments
|
||||
|
||||
1. **Kernel: release device/port/IRQ claims and MSI vectors on death** — the
|
||||
cleanup half of iron rule 1, and the prerequisite for any restart story. Test:
|
||||
kill a claiming driver, spawn it again, the claim succeeds.
|
||||
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||
first subscriber — releasing a dead client's handles is its proof test.
|
||||
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||
`runtime.process` grows the interface above; the service harness handles
|
||||
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||
|
||||
[device-manager.md](device-manager.md) builds directly on all four.
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
- **Signal numbering is not ABI**: the runtime is the stable interface; the numbers
|
||||
beneath it are a private kernel ↔ runtime contract (see "The stable interface").
|
||||
- **Liveness is a `ping` call, not a signal**: signals are statements, questions
|
||||
are synchronous calls (see "Statements, not questions"). A service wanting *deep*
|
||||
health ("can I reach my hardware?") defines its own protocol message on top.
|
||||
- **Process handles: deferred.** Pids + the supervisor gate cover everything
|
||||
planned; transferable handles (Fuchsia-style, delegating signalling without
|
||||
delegating kill) wait for the capability table to grow types beyond endpoints.
|
||||
- **`alarm`: in the vocabulary, unbuilt.** No consumer yet; when one appears it is
|
||||
runtime sugar over the existing timer (arm a timer that posts your own signal) —
|
||||
zero kernel work, so deferring costs nothing.
|
||||
- **Subscription granularity: all exits**, subscriber-side filtering — one
|
||||
subscription per service, a bounded kernel table. Per-id subscriptions only if
|
||||
event volume ever matters (hundreds of processes, not before).
|
||||
- **Client identity across the exit boundary: no convention needed** — an IPC
|
||||
sender's badge already is its task id (see "Who learns of a death").
|
||||
@@ -0,0 +1,120 @@
|
||||
# Process Management
|
||||
|
||||
How danos lists, supervises, and kills processes — the microkernel answer to
|
||||
`ps`, `kill`, and `SIGCHLD`/`wait`.
|
||||
|
||||
## Why system calls, not `/proc`
|
||||
|
||||
Unix systems sit on a spectrum. Classic BSD/macOS list processes through
|
||||
syscalls (`sysctl(KERN_PROC)`) and kill through `kill(2)`; Linux renders the
|
||||
process table as `/proc` for *reading* but still kills through a syscall; Plan 9
|
||||
made the file tree the whole interface (`echo kill > /proc/n/ctl`). Microkernels
|
||||
mostly abandon ambient PIDs: Minix and QNX route everything through a user-space
|
||||
process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||
|
||||
danos rules out `/proc` **as the primitive**: here a `/proc` would be served by
|
||||
the VFS server — a user process — which would put the VFS in the path of process
|
||||
control. If the VFS (or anything under it) hangs, nothing could be listed or
|
||||
killed, *including the hung VFS*. The control plane for processes must not
|
||||
depend on a process. So the primitives are kernel system calls; a read-only
|
||||
`/proc` rendering can be layered on later, and a POSIX-style process-manager
|
||||
server can be built *from* these primitives when one is needed.
|
||||
|
||||
## The three primitives
|
||||
|
||||
### `process_enumerate(buffer, maximum) -> total`
|
||||
|
||||
A snapshot of the task table into a caller buffer of `abi.ProcessDescriptor`
|
||||
(id, supervisor, state, priority, name) — the exact shape of
|
||||
`device_enumerate`, so `ps` is a user program over a snapshot, not a kernel
|
||||
service. The total may exceed what fit; call again with a larger buffer. Kernel
|
||||
tasks are included with an empty name — an honest listing shows the idle tasks
|
||||
too. Ungated and read-only: what is running is not a secret between cooperating
|
||||
bring-up processes.
|
||||
|
||||
### `system_spawn(..., exit_endpoint) -> child id`, and the supervision link
|
||||
|
||||
`system_spawn` records the caller as the child's **supervisor** and returns the
|
||||
child's process id (ids are monotonic, never reused — a stale id can only miss).
|
||||
That link is the kill authority: it answers "who may kill process 7?" without
|
||||
inventing users or permissions, the same way a device *claim* is the capability
|
||||
for `mmio_map`. It composes with the supervision hierarchy the device manager
|
||||
already forms: init supervises the services it starts, the device manager
|
||||
supervises the drivers it matches. (A transferable process *handle* — Fuchsia
|
||||
style — can replace the id once the handle table grows types beyond endpoints.)
|
||||
|
||||
`exit_endpoint` (a handle, or `abi.no_cap`) is the supervisor's death-watch: when
|
||||
the child ends — clean exit, CPU fault, or `process_kill` — the kernel posts an
|
||||
asynchronous notification to that endpoint, exactly like a bound IRQ. The badge
|
||||
carries `abi.notify_badge_bit | abi.notify_exit_bit | child_id`, so one endpoint
|
||||
supervises many children and can even share with IRQ notifications. This is the
|
||||
microkernel's SIGCHLD: no new mechanism, just the IRQ-as-IPC pattern reused, and
|
||||
a supervisor's event loop (`ipc.replyWait`) already knows how to receive it. The
|
||||
child holds a reference to the endpoint from birth, so the notification cannot
|
||||
dangle even if the supervisor dies first.
|
||||
|
||||
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||
|
||||
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||
irrevocable; the exit notification confirms completion.
|
||||
|
||||
## How a kill lands (the kernel mechanics)
|
||||
|
||||
Everything below runs under the big kernel lock, where task states cannot move.
|
||||
|
||||
- **Target ready or blocked** (not on any core): reaped on the killer's own
|
||||
call. The reap releases what death always releases (IRQ bindings first, then
|
||||
a client the target still owed a reply to is failed with `-EPEER`, IPC handles
|
||||
closed, the exit notification posted last) — plus the unlinking only a
|
||||
*remote* death needs: out of the ready queue, out of an endpoint's sender FIFO
|
||||
(`Task.ipc_wait_endpoint`), out of a receive wait queue (`Task.wait_queue`),
|
||||
and out of any server's owed-reply slot, so nothing ever dequeues a dangling
|
||||
pointer. Destroying the address space is safe because no core can have it
|
||||
loaded: every switch away from a task loads the next task's tables.
|
||||
- **Target running on another core**: it cannot be torn down mid-instruction,
|
||||
so it is condemned (`Task.kill_pending`) and dies at whichever comes first:
|
||||
- its next **system_call entry** — checked before dispatch, so a condemned
|
||||
process cannot spawn, claim, or message anything on its way out;
|
||||
- its core's next **timer tick** — but only when the task is not inside one
|
||||
of its own system calls (`Task.in_system_call`): the tick may have
|
||||
interrupted kernel code mid-operation, where teardown would leak whatever
|
||||
the operation held. User-mode execution is always a safe kill point. The
|
||||
tick-time terminate abandons the interrupt frame exactly like the fault
|
||||
path (the LAPIC is acknowledged before the tick hook runs);
|
||||
- any core's tick finding it **blocked or ready** (it entered a syscall and
|
||||
parked after being condemned) — reaped by the same remote-reap path.
|
||||
|
||||
A pure user-mode spin loop that never makes a system call therefore dies
|
||||
within one tick; nothing a process does can outrun the kill.
|
||||
|
||||
The scheduler stays below the process layer: finishing a kill (IRQ bindings,
|
||||
handles, the notification) is called *up* through two hooks process.zig
|
||||
registers at boot (`terminate_current_hook`, `reap_task_hook`), mirroring how
|
||||
the architecture layer calls up into `tick`.
|
||||
|
||||
## Known gaps (bring-up honesty)
|
||||
|
||||
- ~~Device claims are not released on death~~ Closed (M17.1): every path out of a
|
||||
process releases its device claims alongside its IRQ and MSI bindings
|
||||
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||
records how every process ends — exited, a fault class, or killed — before it
|
||||
posts the exit notification, and the supervisor reads it with
|
||||
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||
isolation break).
|
||||
|
||||
## Tests
|
||||
|
||||
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
||||
notifications), `supervision` (the whole user-side surface via the process-test
|
||||
service: spawn supervised → enumerate → kill blocked and spinning children →
|
||||
notifications → gone), `claim-release` (a killed claim-holder's device is
|
||||
claimable again). See test/qemu_test.py.
|
||||
@@ -0,0 +1,70 @@
|
||||
# The release ISO — flashable boot media
|
||||
|
||||
`zig build release-x86-64` produces **`zig-out/danos-x86-64.iso`**, the file you
|
||||
hand to someone who wants to try danos on a real machine: point
|
||||
[balenaEtcher](https://etcher.balena.io) (or Raspberry Pi Imager, or plain `dd`)
|
||||
at it, flash a USB stick, and boot the stick. The same file also burns to
|
||||
optical media. `zig build check-iso-image` validates it without booting.
|
||||
|
||||
```
|
||||
zig build release-x86-64
|
||||
# Etcher: select danos-x86-64.iso → select the stick → Flash
|
||||
# or: sudo dd if=zig-out/danos-x86-64.iso of=/dev/rdiskN bs=4m (macOS; triple-check N)
|
||||
```
|
||||
|
||||
## Why an ISO when danos-usb.img already boots
|
||||
|
||||
`danos-usb.img` is a raw FAT32 **superfloppy** — a filesystem starting at
|
||||
sector 0, no partition table. UEFI firmware accepts that from a USB stick (it
|
||||
probes whole-disk FAT before giving up), which is why `dd`-ing the .img works
|
||||
and why QEMU and the test harness boot it directly. But it is a
|
||||
developer-shaped artifact: flashing apps expect an ISO, and a superfloppy
|
||||
can't be burned to a CD/DVD or carry a partition table for pickier firmware.
|
||||
|
||||
The ISO wraps that same FAT image — bit-identical, built by the same
|
||||
`tools/make-fat-image.py` — in a container that boots everywhere release media
|
||||
gets consumed. One payload, two images: the .img stays the raw volume the QEMU
|
||||
harness mounts and boots, the .iso is what leaves the building.
|
||||
|
||||
## How a hybrid ISO boots twice
|
||||
|
||||
The trick (the same one Linux distribution ISOs use, usually via `xorriso
|
||||
-isohybrid…`) is that ISO9660 reserves its first 32 KiB as a **system area** it
|
||||
never touches — exactly where an MBR lives on a disk. So one file can carry two
|
||||
tables of contents, both pointing at the same embedded FAT image:
|
||||
|
||||
* **Flashed to USB (Etcher, dd):** firmware sees a disk whose sector 0 is an
|
||||
MBR with one partition of type `0xEF` (EFI System Partition) covering the
|
||||
embedded FAT image. It mounts that ESP and runs `\EFI\BOOT\BOOTX64.efi` —
|
||||
the standard removable-media path ([efi.md](efi.md)).
|
||||
* **Burned to optical media:** firmware reads the ISO9660 volume descriptors
|
||||
at sector 16 and finds an **El Torito** boot record. Its catalog has one
|
||||
entry, platform ID `0xEF` (EFI), whose start LBA is — again — the embedded
|
||||
FAT image. The firmware exposes that image as a virtual disk and runs the
|
||||
same `BOOTX64.efi` off it.
|
||||
|
||||
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||
not floppy emulation.
|
||||
|
||||
One El Torito wrinkle: the catalog's sector-count field is 16-bit (units of
|
||||
512 bytes), so it can name at most 32 MiB — less than the 64 MiB FAT image.
|
||||
That is fine in practice: firmware sizes the FAT filesystem from its own BPB,
|
||||
and the boot files sit in the first few MiB of the image (clusters are
|
||||
allocated from the front) either way. The USB path has no such cap.
|
||||
|
||||
## The builder
|
||||
|
||||
`tools/make-iso-image.py` follows the house rule of
|
||||
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||
partition and the El Torito catalog agree on where the FAT image lives and
|
||||
that a FAT32 boot sector is actually there. Every timestamp field in the ISO
|
||||
is zeroed, so the build is reproducible byte-for-byte.
|
||||
|
||||
The ISO9660 filesystem around the boot machinery is minimal but real: a root
|
||||
directory listing `BOOT.CAT` (the catalog) and `EFI.IMG` (the FAT image), so
|
||||
`file`, mount tools, and archive browsers can open the ISO and see what's in
|
||||
it.
|
||||
+16
-3
@@ -1,6 +1,17 @@
|
||||
# Resilience: fault isolation and live restart
|
||||
|
||||
A design/research note, not built yet. This is the property danos is really chasing:
|
||||
Steps 1–4 of the ordering below are **built** (M17–M18, 2026-07-13): user-mode
|
||||
isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||
more of the system moved into restartable processes (the discovery migration,
|
||||
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
@@ -111,9 +122,11 @@ Honest boundaries:
|
||||
## Suggested ordering
|
||||
|
||||
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
||||
path for everything else).
|
||||
path for everything else). **Done.**
|
||||
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
||||
"confine to the process and report it."
|
||||
"confine to the process and report it." **Done** (the kill and reclaim; the
|
||||
supervisor notification waits for step 3's supervisor). A killed server's
|
||||
pending client is unblocked with `-EPEER` rather than hung.
|
||||
3. **A minimal supervisor server** that can (re)start a process.
|
||||
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
||||
table.
|
||||
|
||||
+2
-2
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
|
||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||
|
||||
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task
|
||||
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)).
|
||||
The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
|
||||
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
|
||||
|
||||
## Tasks
|
||||
|
||||
|
||||
+10
-3
@@ -131,7 +131,7 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si
|
||||
model you already have (the interrupt-flag discipline in
|
||||
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
||||
instead of rethinking every critical section. **Done** — see
|
||||
`src/kernel/sync.zig`.
|
||||
`system/kernel/sync.zig`.
|
||||
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
||||
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
||||
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
||||
@@ -151,7 +151,7 @@ next lands.
|
||||
- **Core enumeration** — the MADT parse records every usable Local APIC (with its
|
||||
`apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See
|
||||
[discovery.md](discovery.md).
|
||||
- **The big kernel lock** (`src/kernel/sync.zig`) — one coarse spinlock guarding the
|
||||
- **The big kernel lock** (`system/kernel/sync.zig`) — one coarse spinlock guarding the
|
||||
scheduler queues and IPC, always held with local interrupts disabled. It is held
|
||||
*across* a context switch and released by whichever task resumes (the hand-off
|
||||
rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and
|
||||
@@ -164,7 +164,7 @@ next lands.
|
||||
the highest-priority ready task; per-core queues are a later optimisation.
|
||||
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||
real mode at a low page and runs the [trampoline](../src/kernel/arch/x86_64/trampoline.s)
|
||||
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||
all four cores report `online`.
|
||||
@@ -203,6 +203,13 @@ next lands.
|
||||
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||
BSP's IST stack is static, because it must exist before the frame allocator does.
|
||||
This kept the kernel image small (a static `[128][16 KiB]` IST array would have been
|
||||
2 MiB of `.bss`); it's a few tens of KiB instead.
|
||||
- **`single_threaded` off** — the kernel was built `single_threaded = true`, which
|
||||
compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it
|
||||
quietly breaks the big kernel lock across cores; it's now `false`.
|
||||
|
||||
+13
-1
@@ -1,5 +1,15 @@
|
||||
# System Calls
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
|
||||
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||
|
||||
## The Mechanism of a Syscall
|
||||
|
||||
@@ -37,6 +47,8 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
||||
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
||||
3. **`Yield()`/`Thread_Ctrl()`**
|
||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
|
||||
* * * * *
|
||||
|
||||
|
||||
@@ -0,0 +1,227 @@
|
||||
# System Requirements
|
||||
|
||||
Minimum and recommended hardware for running danos. Every requirement below is
|
||||
grounded in what the current code actually assumes at boot — this is a
|
||||
description of the real target, not an aspirational one.
|
||||
|
||||
## Summary
|
||||
|
||||
danos targets a **modern UEFI x86-64 PC with ACPI and PCIe**. The practical
|
||||
minimum is:
|
||||
|
||||
- 64-bit x86-64 CPU with SSE2, APIC, and `syscall`/`sysret`
|
||||
- UEFI firmware (no BIOS / legacy boot)
|
||||
- ACPI tables: MADT, MCFG, FADT
|
||||
- PCIe with an ECAM (MMConfig) window
|
||||
- **128 MiB RAM** (target); see [Memory](#memory) for the breakdown
|
||||
- USB via **xHCI only**
|
||||
|
||||
There is no support for legacy BIOS boot, x2APIC, port-IO PCI configuration, or
|
||||
any USB host controller other than xHCI.
|
||||
|
||||
## Plain-language hardware guide
|
||||
|
||||
If you don't want to cross-reference chipset datasheets, here's roughly what era
|
||||
of PC works. These are **guidance based on when the required features became
|
||||
standard**, not a list of tested machines — the authoritative rules are in the
|
||||
technical sections below.
|
||||
|
||||
The feature that sets the floor is **built-in xHCI USB** (danos supports no other
|
||||
USB controller) combined with **UEFI firmware**. Both became standard on
|
||||
mainstream desktops and laptops around **2012**.
|
||||
|
||||
| | Known-good baseline | Comfortable recommendation |
|
||||
|---|---|---|
|
||||
| **Intel** | 3rd-gen Core "Ivy Bridge" (2012) with a 7-series "Panther Point" chipset — Intel's first chipset with xHCI built in | 6th-gen Core "Skylake" (2015) or newer |
|
||||
| **AMD** | A-series "Llano" APU with an A75 FCH (2011) — the industry's first chipset with built-in xHCI | Any AM4 platform, i.e. Ryzen (2017) or newer |
|
||||
|
||||
**AMD is not behind Intel here — it was first.** AMD's A75 FCH shipped with
|
||||
native xHCI in April 2011, about a year *ahead* of Intel's 7-series (2012); AMD
|
||||
was the first vendor to earn USB-IF certification for chipset-level USB 3.0. The
|
||||
two "comfortable recommendation" dates differ only because they name convenient,
|
||||
long-supported product lines (Skylake, Ryzen) — not because of any USB
|
||||
capability gap. Every AMD desktop platform from the A75 FCH (2011) and FM2/AM3+
|
||||
era onward has built-in xHCI, and any of them qualifies as a baseline.
|
||||
|
||||
Older 64-bit machines (e.g. Intel Core 2, Nehalem, Sandy Bridge) meet the CPU
|
||||
requirements but typically **lack built-in xHCI and/or ship with BIOS instead of
|
||||
UEFI**, so they are not supported.
|
||||
|
||||
### Matching your CPU by name
|
||||
|
||||
If you know your chip's marketing name or codename, find it here. Everything from
|
||||
the **Supported** rows down works; the **Too old** row does not.
|
||||
|
||||
**Intel Core** (the "-lake"/"-bridge"/"-well" codenames):
|
||||
|
||||
| Status | Generation | Codename(s) | Year |
|
||||
|---|---|---|---|
|
||||
| Too old | 2nd gen | Sandy Bridge | 2011 |
|
||||
| Supported (baseline) | 3rd gen | Ivy Bridge | 2012 |
|
||||
| Supported | 4th–5th gen | Haswell, Broadwell | 2013–2014 |
|
||||
| **Recommended** | 6th–9th gen | **Skylake**, Kaby Lake, Coffee Lake | 2015–2018 |
|
||||
| Recommended | 10th–11th gen | Comet Lake, Ice Lake, Tiger Lake, Rocket Lake | 2019–2021 |
|
||||
| Recommended | 12th gen+ | Alder Lake, Raptor Lake | 2021–2023 |
|
||||
| Recommended | Core Ultra | Meteor Lake, Arrow Lake, Lunar Lake | 2023+ |
|
||||
|
||||
**AMD:**
|
||||
|
||||
| Status | Family | Codename(s) | Year |
|
||||
|---|---|---|---|
|
||||
| Supported (baseline) | A-series APU (A75/A85 FCH) | Llano, Trinity, Richland, Kaveri | 2011–2014 |
|
||||
| Supported | FX (AM3+) | Bulldozer, Piledriver | 2011–2012 |
|
||||
| **Recommended** | **Ryzen** 1000–5000 (AM4) | Summit/Pinnacle Ridge, Matisse, Vermeer (Zen–Zen 3) | 2017–2020 |
|
||||
| Recommended | Ryzen 7000+ (AM5) | Raphael, Granite Ridge (Zen 4 / Zen 5) | 2022+ |
|
||||
| Recommended | Threadripper / EPYC | Zen and later | 2017+ |
|
||||
|
||||
(These map generations to the era their platforms shipped built-in xHCI + UEFI;
|
||||
they are guidance, not a tested-hardware list.)
|
||||
|
||||
**Two caveats that matter regardless of CPU:**
|
||||
|
||||
- **Firmware must be UEFI.** Many 2011-era machines could do either UEFI or
|
||||
legacy BIOS — danos needs it set to UEFI. There is no BIOS boot path.
|
||||
- **Input is PS/2 only, for now.** danos does not yet support USB
|
||||
keyboards/mice. This is fine on most **laptops** (their built-in keyboards are
|
||||
wired to a PS/2-style i8042 controller) but means a **desktop with only USB
|
||||
ports** currently has no usable keyboard. USB HID input is planned.
|
||||
|
||||
Virtual machines are the easiest way to meet every requirement: QEMU (with OVMF/
|
||||
UEFI, a `qemu-xhci` controller, and the default Q35 machine type), or any
|
||||
hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
## CPU / architecture
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:285`, `boot/efi.zig:418` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:282`, `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:59`, `isr.s:169` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:62`, `apic.zig:414` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:279`, `apic.zig:84` |
|
||||
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:144` |
|
||||
|
||||
## Firmware / boot
|
||||
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build.zig:464`, `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
(`efi.zig:578`, `boot-handoff.zig:144`)
|
||||
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||
(`system/devices/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel`, `/system/services/init`, and
|
||||
`/boot/initial-ramdisk.img` off the FAT boot volume. The kernel can boot
|
||||
"kernel-only" without init or the ramdisk. (`efi.zig:14`, `efi.zig:66`)
|
||||
|
||||
## Interrupt controller
|
||||
|
||||
- **Local APIC + I/O APIC required.** I/O APIC base, GSI base, and MADT
|
||||
interrupt-source overrides come from ACPI. (`cpu.zig:365`, `apic.zig:119`)
|
||||
- **MSI supported** — edge-triggered, keyed by vector, no I/O APIC mask cycle.
|
||||
Vector window 33–46, timer on 32, spurious on 47. (`system/kernel/irq.zig:70`,
|
||||
`cpu.zig:397`)
|
||||
- The legacy 8259 PIC is remapped and masked **only if present** (MADT
|
||||
`PCAT_COMPAT`); it is not required. (`apic.zig:103`)
|
||||
|
||||
## PCI / PCIe
|
||||
|
||||
- **PCIe with ECAM (MMConfig) required.** The PCI bus driver maps the host
|
||||
bridge's ECAM window (1 MiB config space per bus) and computes config
|
||||
addresses directly. **There is no legacy CF8/CFC port-IO config path** — the
|
||||
driver bails if the bridge exposes no ECAM window. The ECAM base comes from
|
||||
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:41`, `acpi.zig:6`)
|
||||
|
||||
## USB
|
||||
|
||||
- **xHCI only.** The sole USB driver is `usb-xhci-bus`, and the device manager
|
||||
binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI / OHCI / EHCI exist only
|
||||
as report strings with no driver behind them — **USB 1.x/2.0-only controllers
|
||||
are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||
`system/services/device-manager/device-manager.zig:34`)
|
||||
- USB input (keyboard/mouse over HID) is future work; the current input stack is
|
||||
PS/2. See [Buses & devices](#buses--devices).
|
||||
|
||||
## Timers
|
||||
|
||||
Calibration prefers, in order: (1) CPUID leaf `0x15` TSC frequency, (2) HPET,
|
||||
(3) ACPI PM timer (3.579545 MHz, from FADT), (4) legacy PIT. Any one suffices —
|
||||
HPET/PM-timer/PIT are optional fallbacks when CPUID `0x15` is absent.
|
||||
(`apic.zig:180`)
|
||||
|
||||
- **TSC** — monotonic high-resolution clock.
|
||||
- **LAPIC timer** — scheduler heartbeat, periodic at `timer_hz = 1000 Hz`.
|
||||
(`parameters.zig:39`)
|
||||
|
||||
## Memory
|
||||
|
||||
**Target: 128 MiB RAM.** The system uses 4 KiB pages and a bitmap physical-frame
|
||||
allocator built from the firmware memory map. There is no hardcoded minimum-RAM
|
||||
constant — the allocator only panics if there is no usable region, or none large
|
||||
enough to hold its own bitmap. (`system/kernel/pmm.zig:13`, `pmm.zig:77`)
|
||||
|
||||
Where the budget goes:
|
||||
|
||||
| Consumer | Size | Source |
|
||||
|---|---|---|
|
||||
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:26` |
|
||||
| Kernel stack, per CPU | 16 KiB | `parameters.zig:26` |
|
||||
| IST stack, per CPU | 16 KiB | `parameters.zig:36` |
|
||||
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:32` |
|
||||
| Max concurrent tasks | 32 | `parameters.zig:23` |
|
||||
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:299` |
|
||||
|
||||
The 64 MiB heap cap plus kernel image, per-CPU stacks, task stacks, the frame
|
||||
bitmap, and DMA-contiguous allocations fit comfortably within 128 MiB on a
|
||||
single- or low-core-count machine. Very high core counts (toward the 128-CPU
|
||||
ceiling) add per-CPU stack overhead and push toward more RAM.
|
||||
|
||||
**Note on the 4 GiB physmap:** the loader identity-maps and physmaps the low
|
||||
4 GiB of address space with 2 MiB leaves. This is *virtual address* reach, not a
|
||||
RAM requirement — RAM above 4 GiB simply needs an extra mapping window and is not
|
||||
needed to boot. (`efi.zig:305`)
|
||||
|
||||
Virtual-memory layout (`boot-handoff.zig:47`):
|
||||
|
||||
| Region | Base |
|
||||
|---|---|
|
||||
| User space | `0x0000_7000_0000_0000` |
|
||||
| Kernel heap | `0xFFFF_8000_0000_0000` |
|
||||
| Physmap | `0xFFFF_8800_0000_0000` |
|
||||
| Kernel image | `0xFFFF_FFFF_8000_0000` |
|
||||
|
||||
## Buses & devices
|
||||
|
||||
Buses with real drivers today:
|
||||
|
||||
- **PCIe** via ECAM (`pci-bus`)
|
||||
- **xHCI USB** (`usb-xhci-bus`)
|
||||
- **PS/2** keyboard + mouse (`ps2-bus`) — the current input stack
|
||||
- **Serial UART** (16550/16450), configured from the ACPI SPCR table
|
||||
|
||||
**No storage driver exists yet.** AHCI / NVMe / IDE are named for reporting only;
|
||||
there is no block-device driver. Persistent storage is future work.
|
||||
|
||||
## IOMMU
|
||||
|
||||
**Detection only; enforcement deferred.** The ACPI DMAR table is parsed for the
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInfo`
|
||||
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
|
||||
## What is explicitly NOT supported
|
||||
|
||||
- Legacy BIOS / multiboot / limine boot
|
||||
- 32-bit x86
|
||||
- x2APIC
|
||||
- Legacy port-IO (CF8/CFC) PCI configuration
|
||||
- Non-xHCI USB (UHCI / OHCI / EHCI)
|
||||
- Machines without ACPI (no device discovery)
|
||||
- Persistent storage (no AHCI / NVMe / IDE driver yet)
|
||||
- USB HID input (PS/2 only for now)
|
||||
+36
-4
@@ -1,6 +1,6 @@
|
||||
# SysV: the kernel's calling convention
|
||||
|
||||
Several places in danos say "the kernel is SysV" — most visibly `src/root.zig`:
|
||||
Several places in danos say "the kernel is SysV" — most visibly `system/boot-handoff.zig`:
|
||||
|
||||
```zig
|
||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||
@@ -52,19 +52,51 @@ argument arrives in **RCX**, not RDI.
|
||||
|
||||
danos's two binaries default to different conventions:
|
||||
|
||||
- `src/boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||
- `boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||
Microsoft x64 (first argument → RCX).
|
||||
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
||||
|
||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||
garbage. So both sides reference the same `danos.kernel_abi` (SysV): the loader's
|
||||
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||
function-pointer type and the kernel's `_start` both carry
|
||||
`callconv(danos.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||
the handoff it governs.
|
||||
|
||||
## The process-entry stack (argc/argv)
|
||||
|
||||
The SysV ABI also fixes what a *fresh process* finds on its stack — and danos
|
||||
follows it, so its own runtime and any future C libc read arguments the same way.
|
||||
At the first user instruction, `rsp` is 16-byte aligned and points at (addresses
|
||||
growing upward):
|
||||
|
||||
```
|
||||
rsp → argc u64
|
||||
argv[0] … argv[argc-1] pointers into the strings area below
|
||||
NULL argv terminator
|
||||
NULL envp terminator (no environment yet)
|
||||
{AT_PAGESZ, page size} auxiliary vector
|
||||
{AT_NULL, 0} auxiliary-vector terminator
|
||||
argv string bytes NUL-terminated
|
||||
───────────────────────── stack top (stack_top_virtual)
|
||||
```
|
||||
|
||||
The kernel builds this block at the top of the process's stack — 8 pages (32 KiB,
|
||||
`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below
|
||||
them left unmapped as a **guard**, so a stack overflow faults (killing only that
|
||||
process) instead of silently corrupting the image
|
||||
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`runtime.process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||
accepted). A C runtime's `crt0` would walk
|
||||
the identical layout unmodified — that's the compatibility being bought. The
|
||||
`args` test proves the round trip.
|
||||
|
||||
## Where else it surfaces
|
||||
|
||||
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
||||
|
||||
+16
-6
@@ -8,8 +8,9 @@ without a human staring at the screen.
|
||||
There are two layers:
|
||||
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||
the shared `danos` module (the handoff layout in `src/root.zig`). These compile
|
||||
for the host and run natively.
|
||||
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||
stays self-consistent. These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
|
||||
@@ -17,7 +18,7 @@ There are two layers:
|
||||
|
||||
The framebuffer console draws pixels, which a test can't read without
|
||||
screen-scraping. So the kernel also writes everything to a **serial port**
|
||||
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
||||
appears on serial as plain text.
|
||||
|
||||
@@ -26,10 +27,19 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||
so a serial-enabled image is still safe on real hardware.)
|
||||
|
||||
## In-kernel test cases
|
||||
|
||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured
|
||||
self-test from `system/kernel/tests.zig` instead of idling. Each case writes structured
|
||||
markers to serial:
|
||||
|
||||
```
|
||||
@@ -108,7 +118,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
|
||||
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
||||
one — means:
|
||||
|
||||
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||
1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||
tables) behind the same `arch` interface,
|
||||
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
||||
|
||||
@@ -118,7 +128,7 @@ architectures".
|
||||
|
||||
## Writing a new case
|
||||
|
||||
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||
1. Add a function to `system/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
||||
or trigger the condition and rely on the handler's output (faulting cases).
|
||||
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
||||
|
||||
@@ -0,0 +1,469 @@
|
||||
# Threading — build plan (`runtime.Thread` over a private thread ABI)
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **`runtime.Thread` mirrors `std.Thread`'s API; the implementation is danos-native.**
|
||||
Not literal `std.Thread` — that would break the [private ABI](syscall.md).
|
||||
- **Threads are a narrow, per-binary opt-in.** Default concurrency stays process + IPC
|
||||
([resilience.md](resilience.md)); only a service that asks is built
|
||||
`single_threaded = false`.
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
supervisor restarts the process, which respawns its threads.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan runs unattended). A
|
||||
thread proves it ran by writing to **shared memory** the parent reads back, and proves
|
||||
parallelism by stamping the **core index** it ran on (like the `smp`/`affinity` cases).
|
||||
|
||||
- `zig build test` — host unit tests (closure packing, mutex state machine, futex
|
||||
wrapper encodings).
|
||||
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial
|
||||
markers. Thread cases set `smp: true` (real parallelism) and bump `mem` (they boot
|
||||
the process/scheduler stack); each milestone **adds its case to `CASES`** so its gate
|
||||
is runnable.
|
||||
- **Guardrail every milestone:** the concurrency-sensitive existing cases stay green —
|
||||
`smoke`, `sched`, `priority`, `smp`, `affinity`, `process`, `process-kill`,
|
||||
`supervision`, `fault-recovery`, `vfs-client-death`, `ipc`/`ipc-cap`,
|
||||
`display-service`. A threading change that regresses those is rejected.
|
||||
|
||||
## Unattended execution (the loop contract)
|
||||
|
||||
This plan runs to completion **without human input**. Every design choice is already
|
||||
fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration must:
|
||||
|
||||
1. **Resume** at the first milestone that still has an unchecked `- [ ]`. (All earlier
|
||||
milestones are done — do not revisit them.)
|
||||
2. **Work on a branch.** On the first iteration, branch off the current `main` into a new
|
||||
branch (e.g. `threading-phase2` — Phase 1's `threading` is already merged); never
|
||||
commit to `main` directly. Push that **branch** to `origin` after each milestone (step
|
||||
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||
`main` stays a human step.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||
set**, then `zig build` (clean) and `zig build test` (green).
|
||||
5. **Decide, do not ask:**
|
||||
- **Green** = the milestone's case prints its stated marker(s) and reports `PASS`,
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
(`zig-out/qemu-test/<case>-failed-serial.log`) and fix in place, then re-run — up to
|
||||
**3 fix attempts** for that gate. A concurrency case that fails then passes on a
|
||||
bare re-run is **flaky, not green**: re-run it **twice more** and treat green only
|
||||
if it passes all; otherwise fix the race (a real threading bug), don't paper over
|
||||
it.
|
||||
6. **A genuinely ambiguous fork is not a stop.** Pick the option most consistent with
|
||||
[threading.md](threading.md)'s *Locked decisions*, note the choice in the commit
|
||||
message, and continue. Do not pause for confirmation on in-scope, reversible work —
|
||||
this plan is that authorization.
|
||||
|
||||
**The only stop conditions:**
|
||||
|
||||
- **Done** — every milestone box **in this plan** is checked (M1 through M11), `zig build`
|
||||
clean, the whole `thread-*` suite + guardrail green. Phase 1 (M1–M6) is *already*
|
||||
checked, so do **not** read that as Done: the loop's real work is the first plan section
|
||||
that still has unchecked boxes — Phase 2 (M7–M11). Only stop when M7–M11 are all checked
|
||||
too. Update threading.md's status line, push the final branch state to `origin`, and
|
||||
stop. The branch is on `origin` for review; **merging Phase 2 into `main` is the user's
|
||||
step**, not the loop's.
|
||||
- **Blocked** — a gate is still red after 3 fix attempts, or a step needs something
|
||||
outside the repo (a toolchain change, new hardware, a decision no locked decision
|
||||
covers). Append `> **BLOCKED (M<n>):** <what failed, what was tried, the serial
|
||||
marker missing>` under that milestone, commit **and push** the WIP on the branch, and
|
||||
stop. Do not thrash further and do not silently skip the milestone.
|
||||
|
||||
Nothing else warrants stopping — not "should I proceed?", not "is this right?". The
|
||||
checkboxes + git history are the resumable record; the next iteration picks up from the
|
||||
first unchecked box.
|
||||
|
||||
---
|
||||
|
||||
## M1 — Address-space refcount (kernel foundation, no API, no behaviour change) ✅
|
||||
|
||||
The one invariant change threads require, landed and proven **before** anything shares
|
||||
an address space. Today address space is 1:1 with a task and teardown destroys it on any user
|
||||
task's exit; make destruction happen on the **last** exit.
|
||||
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
- [x] `-Dtest-case=address-space-refcount`: spawn and reap several ring-3 processes in sequence
|
||||
and assert (via test-observable `liveAddressSpaceCount`/`addressSpaceDestroyCount`) that the
|
||||
live-space count returns to **baseline** and destructions advance by exactly that
|
||||
many — each space destroyed exactly once, no leak, no double-free. (Refcount
|
||||
observables, not raw frame counts, since kernel stacks are still leaked on exit.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py address-space-refcount` passes
|
||||
(`address-space-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the
|
||||
full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `smp`,
|
||||
`affinity`, `process`, `process-kill`, `supervision`, `fault-recovery`,
|
||||
`vfs-client-death`, `ipc`, `ipc-cap`, `display-service`); default `zig build` clean,
|
||||
`zig build test` green. The reframing is invisible until an address space is actually shared.
|
||||
|
||||
## M2 — `thread_spawn` + `thread_exit`: a thread runs in the shared address space ✅
|
||||
|
||||
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||
space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's
|
||||
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
(`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new
|
||||
thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a
|
||||
process) — no naked runtime asm.
|
||||
- [x] `library/runtime/thread.zig` (barrel-exported as `runtime.Thread`): `spawn` maps a
|
||||
stack (`mmap`), heap-allocates the `{args}` closure, and calls
|
||||
`thread_spawn(&Closure.entry, stack_top, closure)`; `Closure.entry` (a plain C-ABI
|
||||
Zig fn, closure in rdi) runs the function and calls `thread_exit`. Stack top is
|
||||
16-aligned-minus-8 for the C entry.
|
||||
- [x] A `threaded` flag on the user-binary recipe (`addThreadedUserBinary` →
|
||||
`single_threaded = false`); `thread-test` is the first opt-in binary.
|
||||
- [x] `-Dtest-case=thread-spawn`: `thread-test` spawns a worker that writes a sentinel to
|
||||
a **shared** global and release-stores `done`; the main thread acquire-polls `done`
|
||||
and asserts the shared global holds the sentinel — proof the worker ran in the same
|
||||
address space.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-spawn` passes
|
||||
(`thread-test: child ran in shared address space ok` → `DANOS-TEST-RESULT: PASS`); guardrail set
|
||||
16/16 green (incl. `args`/`init`/`process`, which exercise the new `jump_to_user_arg`
|
||||
process path with arg 0) plus `address-space-refcount`; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Note (deferred to M3+):** the mmap arena is per-*task* (`heap_next`), so two threads
|
||||
> in one address space that both `mmap` would collide. Fine for M2 (only the parent maps, for the
|
||||
> child's stack); make the arena per-address-space and the runtime heap thread-safe alongside the
|
||||
> `Mutex` work (M5).
|
||||
|
||||
## M3 — `join` + `detach` + real parallelism ✅
|
||||
|
||||
- [x] `join` over the existing exit-notification path
|
||||
([process-lifecycle.md](process-lifecycle.md)): `thread_spawn` gained a 4th arg, an
|
||||
`exit_endpoint` handle (resolved + refcounted like `spawnProcessSupervised`, via
|
||||
`spawnThreadSupervised`); `join` blocks in `ipc_reply_wait` on that endpoint until
|
||||
the child-exit notice for its `tid`, then `munmap`s the stack. `detach` relinquishes
|
||||
the join right (its stack is reclaimed at process exit — kernel-reaper reclaim for
|
||||
detached threads is deferred; see note).
|
||||
- [x] `runtime.Thread.join` / `detach`, plus `Thread.currentCore()` (a new `current_core`
|
||||
= 39 syscall) for the parallelism proof. `getCurrentId` deferred to M6 (TLS), where
|
||||
a lighter self-id fits. The closure now rides the **thread's own stack** (not the
|
||||
heap) — private per thread, so spawn/join touch no shared heap.
|
||||
- [x] `-Dtest-case=thread-join` (`smp: 4`): `thread-test` join mode spawns N=4 workers
|
||||
that each do K=100k `@atomicRmw`-increments on a shared counter and stamp the core
|
||||
they ran on; the main thread joins all N and asserts `counter == N*K` **and**
|
||||
`@popCount(cores_seen) > 1` (genuine cross-core parallelism), then a detached worker
|
||||
proves `detach` runs without a join.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-join` passes (`thread-test: join ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 4 runs; guardrail 17/17 green (incl. `smp`,
|
||||
`affinity`, `process-kill`, and `args`/`init`/`process` on the exit-endpoint spawn path)
|
||||
plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note (deferred):** a detached thread's stack is freed only at process exit (not by the
|
||||
> reaper on thread exit) — kernel user-stack tracking + reclaim is a later refinement. And
|
||||
> the runtime heap is still not thread-safe: threads that both allocate concurrently would
|
||||
> race (the thread *machinery* avoids the heap, but worker code sharing an allocator does
|
||||
> not). Both fold into the M5 `Mutex`/allocator work.
|
||||
|
||||
## M4 — Futex: the one blocking primitive ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||
count)` scans the task table and readies up to `count` matching waiters (same
|
||||
address space). No spinning — a parked waiter leaves its core free to `hlt`. A
|
||||
timed wait also sets `wake_at`, so the timer's `wakeExpired` wakes it; `futex_addr`
|
||||
staying non-zero (only `futex_wake` clears it) is how the waiter tells timeout from
|
||||
a real wake.
|
||||
- [x] `runtime.Thread.Futex` (`wait` / `timedWait` / `wake`) over the syscall wrappers.
|
||||
- [x] `-Dtest-case=thread-futex` (`smp: 4`): a waiter thread prints `waiting` and
|
||||
`futex_wait`s on a word; the main thread publishes it, prints `waking`, and
|
||||
`futex_wake`s; the waiter prints `woke`. Then a `timedWait` on an unwoken word
|
||||
reports `error.Timeout`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-futex` passes, robust across 3 runs —
|
||||
the case's **ordered** regex asserts `waiting → waking → woke → PASS` on the serial
|
||||
stream (the handoff proof), and `thread-futex: timeout ok` confirms the timeout.
|
||||
Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `address-space-refcount`,
|
||||
`thread-spawn`, `thread-join`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note:** the kernel test checks only the freshest verdict marker via `bufferHas` (the
|
||||
> in-memory log ring buffer evicts older lines); ordering is asserted against the full
|
||||
> serial stream by the qemu regex instead.
|
||||
|
||||
## M5 — `Mutex` + `Condition` + `Semaphore` ✅
|
||||
|
||||
- [x] `runtime.Thread.Mutex` (three-state futex mutex: CAS fast path, `futex_wait`/`wake`
|
||||
slow path), `Condition` (`wait`/`timedWait`/`signal`/`broadcast`, a futex sequence
|
||||
counter), `Semaphore` (permits over `Mutex`+`Condition`) — the same state machines
|
||||
`std.Thread` uses, ported onto our `Futex`.
|
||||
- [x] `-Dtest-case=thread-mutex` (`smp: 4`): a bounded producer/consumer — 2 producers +
|
||||
2 consumers over one `Mutex` and two `Condition`s move N=2000 unique items through
|
||||
an 8-slot ring; the consumed checksum and tally match exactly (no lost/duplicated
|
||||
item, no overrun) under real cross-core contention. The small ring forces producers
|
||||
to block on full and consumers on empty, exercising `Condition.wait`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-mutex` passes (`thread-mutex: ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 3 runs; guardrail 17/17 green (incl.
|
||||
`sleep`/`event`/`ipc`) + all M1–M4 thread cases; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Deferred (with rationale):**
|
||||
> - **`join` → futex completion word** — the exit-endpoint join (M3) is correct and
|
||||
> tested. A futex-completion join needs the *kernel* to clear+wake a word after the
|
||||
> thread is fully off its stack (a CLONE_CHILD_CLEARTID-style mechanism); doing it in
|
||||
> the thread's own trampoline would let `join` `munmap` the stack while the thread still
|
||||
> runs on it (use-after-free). Left on the exit-endpoint path; the kernel clear-on-exit
|
||||
> is a later, separate refinement.
|
||||
> - **Host unit tests for the state machines** — `Mutex`/`Condition` bottom out in the
|
||||
> `futex_*` syscalls, unavailable on the host without a mockable `Futex` seam. The QEMU
|
||||
> `thread-mutex` gate exercises them under real concurrency instead; a host-side mock is
|
||||
> future work.
|
||||
|
||||
## M6 — `getCurrentId`, docs, and CI wiring ✅
|
||||
|
||||
- [x] `getCurrentId` via a small `thread_self = 42` syscall (`runtime.Thread.getCurrentId`
|
||||
returns the kernel task id). **Per-thread `threadlocal` TLS is deferred** — no
|
||||
consumer needs it, and it would require context-switching the thread pointer per task
|
||||
(real kernel + per-switch cost) for an unused feature; threaded binaries have run fine
|
||||
without it through M2–M5. threading.md's TLS reasoning already scoped it as
|
||||
deferred-unless-needed. When a consumer appears, the shape is: `thread_spawn`
|
||||
allocates a per-thread TLS block, sets the thread pointer, and the context switch saves/
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||
status updated to **built**; the worked example is threading.md's win-condition.
|
||||
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||
thread confirms all three ids are non-zero and distinct — each thread has its own
|
||||
kernel identity. (Renamed from `thread-tls`, which implied `threadlocal`.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-id` passes; the whole `thread-*` suite
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`) plus the full guardrail set pass; default
|
||||
`zig build` clean, `zig build test` green.
|
||||
|
||||
---
|
||||
|
||||
## Status
|
||||
|
||||
**Phase 1 (M1–M6): built.** danos has `runtime.Thread` — `spawn`/`join`/`detach`,
|
||||
cross-core parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private
|
||||
thread ABI behind the runtime.
|
||||
|
||||
**Phase 2 (M7–M11): built.** Thread-safe allocation (M7), a task reaper that reclaims dead
|
||||
tasks' kernel stacks (M8), endpoint-free `thread_join` (M9), the per-thread thread pointer (M10),
|
||||
and `RwLock`/`WaitGroup` + host-testable sync (M11). Two things stay deferred by design
|
||||
(no consumer): the Zig `threadlocal` *compiler* layer (M10) and detached-thread user-stack
|
||||
reclaim (M9) — both noted in place.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2 — hardening (M7–M11)
|
||||
|
||||
The organising principle, so Phase 2 reinforces danos's goals rather than eroding them:
|
||||
|
||||
- **Everything a thread owns is reclaimed on process death.** Thread stacks, TLS blocks,
|
||||
and futex words live in the process's **address space**, and the kernel's per-process
|
||||
state is keyed by the address-space root — so the M1 refcount + `destroyAddressSpace` already
|
||||
free all of it when the last thread exits. A crashed or killed threaded process leaves
|
||||
**nothing** behind. Phase 2 closes the one thing that is *not* address-space-owned — the
|
||||
per-task **kernel** stack (kernel heap) — with a reaper (M8). This is the
|
||||
[resilience](resilience.md) restart guarantee, extended to threads.
|
||||
- **Kernel owns mechanism; the runtime owns policy.** The kernel maps pages, saves/
|
||||
restores the thread pointer, and reaps dead tasks; the runtime decides allocation, TLS layout,
|
||||
and lock algorithms. Every new kernel entry stays a private syscall behind the runtime
|
||||
([syscall.md](syscall.md)) — the ABI stays renumberable.
|
||||
- **The process is still the isolation and restart boundary.** Threads share fate within
|
||||
one process; Phase 2 never adds a way for one process to reach into another (the
|
||||
cross-process futex stays explicitly out of scope, below).
|
||||
|
||||
### M7 — Thread-safe allocation (the correctness gap) ✅
|
||||
|
||||
Today the mmap arena cursor is per-*task* and the runtime heap is unlocked, so two
|
||||
threads in one process that both allocate corrupt each other. The thread *machinery*
|
||||
avoids this (closure on the stack, stacks mmap'd only by the spawner), but real
|
||||
multi-threaded code would hit it. Closed it:
|
||||
|
||||
- [x] **Kernel — per-address-space mmap arena.** Grew M1's `address_space_refs` entry into the
|
||||
per-address-space object holding the `mmap`/`mmio` arena cursors (moved off `Task`);
|
||||
`scheduler.addressSpaceMmapNextPtr`/`addressSpaceDeviceMapNextPtr` expose them. `systemMmap`
|
||||
reserves a disjoint range under a *brief* lock, then maps **per page** under a
|
||||
short-held lock — not the whole grant — because the big lock is held with interrupts
|
||||
disabled, so pinning it across a multi-MiB memset+map froze other cores (it timed
|
||||
the `affinity` scenario out mid-bring-up). Freed at refcount zero, so the cursors
|
||||
vanish with the process.
|
||||
- [x] **Runtime — thread-safe heap.** The allocator's two free-list mutators
|
||||
(`rawAlloc`/`rawFree`) take a `Thread.Mutex`, gated on
|
||||
`!@import("builtin").single_threaded` so single-threaded binaries compile it out and
|
||||
pay nothing. Uncontended acquisition is a single CAS (no syscall).
|
||||
- [x] `-Dtest-case=thread-alloc` (`smp: 4`): 4 threads each do 500 `alloc`/fill/verify/
|
||||
`free` cycles of varied sizes; each block is filled with a per-thread pattern and
|
||||
verified before free, so any overlap between concurrent allocations is caught.
|
||||
|
||||
**Gate (met):** `thread-alloc` passes (3× non-flaky); full guardrail 23/23 green,
|
||||
`zig build`/`zig build test` clean.
|
||||
|
||||
> **Also fixed here:** the `affinity` guardrail's fixed-count busy-loop (`while (spins <
|
||||
> 3e9)`) had codegen-dependent wall-time — adding a function to `tests.zig` flipped how
|
||||
> the optimiser compiled it, swinging affinity from ~4 s to ~63 s and timing it out.
|
||||
> Reworked it (and the settle loop) to wait on the wall clock instead, so its duration is
|
||||
> independent of unrelated code changes.
|
||||
|
||||
### M8 — The task reaper (cleanup + resilience) ✅
|
||||
|
||||
A dead task's **kernel** stack was leaked ("no reaper yet") — every process *and* thread
|
||||
death lost one, so a crash loop bled kernel memory. The reaper fixes it and serves the
|
||||
[resilience](resilience.md) restart goal directly:
|
||||
|
||||
- [x] A dying task cannot free the kernel stack it runs on, so `exit()`/`exitUserLocked`
|
||||
record it in a **per-core `reap_after_switch` slot** and switch away; the task that
|
||||
resumes on that core frees the stack in `switchTo`'s tail (it's on its own stack, the
|
||||
big lock is still held so the slot can't have been reused). A **tick-time drain**
|
||||
(`reapKillPendingLocked`) is the safety net for the case where the next task is
|
||||
*fresh* (enters via the trampoline, bypassing `switchTo`'s tail). A task killed while
|
||||
*not* running is freed immediately in `destroyTaskLocked`. A `live_stack_bytes`
|
||||
counter is the observable. *(Detached-thread user-stack reclaim moves to M9, which
|
||||
adds the joinable/detached flag.)*
|
||||
- [x] `-Dtest-case=task-reap` (`smp: 4`): spawn and kill 12 processes; poll the
|
||||
test-observable `scheduler.liveStackBytes()` until it returns to **baseline** (a
|
||||
correct reaper gets there in a few ms; a genuine leak times out) — every kernel
|
||||
stack reclaimed, no leak. Threads exit through the same `exitUserLocked`, so covered.
|
||||
|
||||
**Gate (met):** `task-reap` passes (5× isolated + 2× in the full batch); `fault-recovery`,
|
||||
`supervision`, `process-kill`, `address-space-refcount`, `smp`, `affinity` all still green (24/24
|
||||
full guardrail); `zig build`/`zig build test` clean.
|
||||
|
||||
> **Bug found + fixed here (touches every context switch):** the post-`switchContext` reap
|
||||
> first read the `pc` **parameter**, but a task that migrated cores carries a *stale* `pc`
|
||||
> in its saved `switchTo` frame — so it read the wrong core's slot and freed a live stack
|
||||
> (a #GP under SMP). Fixed to re-fetch `thisCpu()` after the switch (the switch only swaps
|
||||
> stacks on the current core).
|
||||
|
||||
### M9 — Futex-completion join (retire the per-thread endpoint)
|
||||
|
||||
With the reaper (M8) able to act *after* a thread is fully off its stack, migrate `join`
|
||||
to the std shape and drop M3's per-thread exit endpoint:
|
||||
|
||||
- [x] A **`thread_join(tid)` syscall** (not a user futex word): it blocks the caller until
|
||||
the task with id `tid` exits, and the exit paths call `wakeJoinersLocked`. `join`
|
||||
only reclaims the joined thread's **user** stack, which the thread vacates the moment
|
||||
it enters the kernel to exit — so waking at *exit* time (not reap time) is safe, and
|
||||
no reaper/address-space juggling or user-memory write is needed. This is equally
|
||||
std-shaped (like `pthread_join`) and much simpler/safer than the planned
|
||||
reaper-written completion word. `thread_spawn` no longer takes an exit endpoint (the
|
||||
runtime passes `no_cap`); the per-thread IPC endpoint is gone.
|
||||
- [x] `thread-join` passes on the new path, and its join mode now runs **40 spawn+join
|
||||
cycles** — under the old per-thread-endpoint scheme those leaked handles would
|
||||
exhaust the 16-slot handle table; here they all succeed, proving join is endpoint-free.
|
||||
|
||||
**Gate (met):** `thread-join` passes (3× isolated) on the `thread_join` path; full
|
||||
guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-reap`);
|
||||
`zig build`/`zig build test` clean.
|
||||
|
||||
> **Reaper hardened here (fixes an M8 flake).** M8's single per-core reap slot could be
|
||||
> *overwritten* by a second death on that core before the first drained (a fresh-task/SMP
|
||||
> timing window) — an intermittent one-stack leak (`task-reap` flaked ~20%). Replaced it
|
||||
> with a per-core reap **list** plus a `.reaping` task state so a pending slot can't be
|
||||
> reused before its stack is freed. `task-reap` now 11/11 isolated + 2× in the batch.
|
||||
|
||||
> **Deferred:** detached-thread **user-stack** reclaim (still freed at process exit, as in
|
||||
> M3). Doing it in the reaper needs the saved address space + stack range and a
|
||||
> translate/unmap in a not-currently-loaded address space — real complexity for a bounded leak.
|
||||
> A follow-up when a consumer needs it.
|
||||
|
||||
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||
|
||||
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||
|
||||
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||
**only when it changes** (the same conditional-load discipline as CR3;
|
||||
`architecture.setThreadPointer` → `wrmsr IA32_FS_BASE` on x86_64). A
|
||||
`set_thread_pointer(addr)` = 44 syscall sets the caller's `thread_pointer` and loads it
|
||||
now. The kernel never touches FS, so there is no swapgs complication.
|
||||
- [x] **Runtime** lays a small per-thread TLS block at the top of each thread's stack
|
||||
(self-pointer at `%fs:0` + scratch slots) and the thread trampoline calls
|
||||
`set_thread_pointer` before any user code — so every spawned thread has a private,
|
||||
switch-stable thread pointer. Reclaimed with the stack.
|
||||
- [x] `-Dtest-case=thread-tls` (`smp: 4`): two threads each write a unique marker to their
|
||||
own `%fs:8` slot and — after both have written — read it back; a shared (non-per-thread)
|
||||
FS base would clobber one and cause cross-talk. Both read their own marker → pass.
|
||||
|
||||
**Gate (met):** `thread-tls` passes (3×); full guardrail 25/25 (the switch-time thread-pointer
|
||||
restore touches every context switch); `zig build`/`zig build test` clean.
|
||||
|
||||
> **Deferred: the Zig `threadlocal` *compiler* layer.** Real `threadlocal` variables need
|
||||
> the ELF **variant-II TLS** surface — `.tdata`/`.tbss` sections + a `PT_TLS` program header
|
||||
> in `user.ld`, a runtime that copies the template with exact negative-offset layout, and
|
||||
> the `.large`-code-model TLS section names — a high-uncertainty lift for a feature with
|
||||
> **no consumer today** (threading.md scopes it "only if a consumer needs it"). What lands
|
||||
> here is the load-bearing piece — the per-thread thread pointer, context-switched — so adding the
|
||||
> compiler layer later is purely runtime+linker work on top, no kernel change. `getCurrentId`
|
||||
> stays the `thread_self` syscall (M6) rather than an fs self-slot (which would need the
|
||||
> main thread's TLS set up in `_start` too).
|
||||
|
||||
**Gate:** `thread-tls` passes; full `thread-*` suite + guardrail green.
|
||||
|
||||
### M11 — `RwLock`, `WaitGroup`, and host-testable sync ✅
|
||||
|
||||
- [x] `runtime.Thread.RwLock` (reader-preferring: `>0` readers / `-1` writer / `0` free,
|
||||
with `lock`/`tryLock`/`unlock` + `lockShared`/`tryLockShared`/`unlockShared`) and
|
||||
`WaitGroup` (`start`/`finish`/`wait`), both on the existing `Mutex`/`Condition`.
|
||||
- [x] A compile-time `Futex` seam gated on `builtin.os.tag == .freestanding`: the futex
|
||||
syscalls on danos, a spin+yield mock off-target (Zig 0.16 has no `std.Thread.Futex`;
|
||||
`wake` is a no-op since the state machines re-check). `thread.zig` is wired into
|
||||
`zig build test`, so `Mutex`/`RwLock`/`WaitGroup` run as **host unit tests** with real
|
||||
`std.Thread` threads (`test` blocks only compile under test).
|
||||
- [x] `-Dtest-case=thread-rwlock` (`smp: 4`): 2 writers set both halves of a value under
|
||||
the exclusive lock while 3 readers check the halves match under the shared lock —
|
||||
zero half-write observations across ~150k reads. Host tests cover the Mutex,
|
||||
RwLock, and WaitGroup state machines.
|
||||
|
||||
**Gate (met):** `zig build test` covers the sync primitives (host threads); `thread-rwlock`
|
||||
passes (3×); full Done gate **26/26** (whole `thread-*` suite + guardrail); `zig build`
|
||||
clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through an [shm](display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
- **Per-thread signal delivery** — signals stay process-scoped
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
swaps the impl under `runtime.Thread`, not the call sites.
|
||||
@@ -0,0 +1,332 @@
|
||||
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||
tasks' kernel stacks. Deferred by design (no consumer yet): the Zig `threadlocal`
|
||||
*compiler* layer (the per-thread thread pointer is in place, so it's runtime+linker work on top) and
|
||||
detached-thread user-stack reclaim — see the plan's M9/M10 notes. The analysis is against
|
||||
**Zig 0.16** (the pinned toolchain); `std.Thread`'s internals move between releases, so
|
||||
treat upstream shapes as "0.16.x."
|
||||
|
||||
## The win condition
|
||||
|
||||
A danos service can write
|
||||
|
||||
```zig
|
||||
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||
// ... do other work concurrently ...
|
||||
t.join();
|
||||
```
|
||||
|
||||
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||
coordination — **without any code path reaching the kernel except through the
|
||||
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||
implementation underneath, not the API above.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
features*; the implementation underneath is danos-native. See
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||
|
||||
## Why not literal `std.Thread`
|
||||
|
||||
danos's ABI invariant is that the **runtime is the sole holder of the syscall ABI**,
|
||||
and that ABI is private and renumberable ([syscall.md](syscall.md) — "unstable
|
||||
private ABI"). That is a security and evolvability asset: no compiled binary can
|
||||
hardcode a syscall number, and the kernel can renumber freely because only the
|
||||
runtime — rebuilt in lockstep — knows the mapping.
|
||||
|
||||
`std.Thread` is incompatible with that invariant on two counts:
|
||||
|
||||
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
|
||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||
invariant.
|
||||
|
||||
## Where threads fit: the resilience tension
|
||||
|
||||
Threads are in genuine tension with a resilience-first microkernel, and it is worth
|
||||
being explicit so we do not reach for them by reflex.
|
||||
|
||||
The reason danos pays for a microkernel is **fault isolation**
|
||||
([resilience.md](resilience.md)): a component corrupts its own address space, faults,
|
||||
and is **restarted** without touching anyone else — because the boundary *is* the
|
||||
address space. Threads deliberately remove that boundary *within* a process:
|
||||
|
||||
- Threads share one address space, so one thread's stray write corrupts them all —
|
||||
there is no isolation **between** threads.
|
||||
- Threads share fate: a fault in any thread, or a "kill the process" decision, takes
|
||||
down **all** of them. Restartability lives at the process level, not the thread
|
||||
level.
|
||||
- Shared mutable state reintroduces data races — the failure class the
|
||||
isolate-and-message model was chosen to avoid.
|
||||
|
||||
**Therefore:** the default answer to "make X concurrent" stays *another process over
|
||||
IPC* (isolated, independently restartable) or a single event loop with several
|
||||
message sources. Reach for a thread only inside **one** service that needs genuine
|
||||
**shared-memory, low-latency parallelism** and can accept intra-service fate-sharing —
|
||||
e.g. a compositor splitting tile compositing across cores, where per-tile IPC would be
|
||||
too chatty. "Input on one thread, display on another" is *not* that case; it wants two
|
||||
processes. The isolation boundary stays at process granularity.
|
||||
|
||||
## The API surface (mirrors `std.Thread`)
|
||||
|
||||
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||
|
||||
```zig
|
||||
pub const Thread = struct {
|
||||
pub const Id = u32; // the kernel task id
|
||||
pub const SpawnConfig = struct {
|
||||
stack_size: usize = default_stack_size,
|
||||
allocator: ?std.mem.Allocator = null, // for the closure + stack bookkeeping
|
||||
};
|
||||
pub const SpawnError = error{ OutOfMemory, ThreadQuotaExceeded, SystemResources };
|
||||
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread;
|
||||
pub fn join(self: Thread) void; // block until the thread ends, reclaim its stack
|
||||
pub fn detach(self: Thread) void; // give up the right to join; kernel reclaims on exit
|
||||
pub fn getCurrentId() Id;
|
||||
pub fn yield() void; // -> existing `yield` syscall
|
||||
|
||||
pub const Mutex = struct { pub fn lock(*Mutex) void; pub fn tryLock(*Mutex) bool; pub fn unlock(*Mutex) void; };
|
||||
pub const Condition = struct { pub fn wait(*Condition, *Mutex) void; pub fn timedWait(*Condition, *Mutex, u64) error{Timeout}!void; pub fn signal(*Condition) void; pub fn broadcast(*Condition) void; };
|
||||
pub const Semaphore = struct { pub fn wait(*Semaphore) void; pub fn post(*Semaphore) void; };
|
||||
pub const Futex = struct { pub fn wait(*const atomic.Value(u32), u32) void; pub fn timedWait(...) error{Timeout}!void; pub fn wake(*const atomic.Value(u32), u32) void; };
|
||||
// RwLock / ResetEvent / WaitGroup follow the same pattern, added as needed.
|
||||
};
|
||||
```
|
||||
|
||||
Deviations from `std.Thread`, called out honestly:
|
||||
|
||||
- **The thread function's return value is discarded** (as `std.Thread.join` returns
|
||||
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||
return.
|
||||
- `getCpuCount()` maps to the existing SMP core count ([smp.md](smp.md)); a service
|
||||
rarely needs it.
|
||||
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Four new entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36`, each with a `library/runtime` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
| `thread_spawn` | `(entry, stack_top, arg) -> tid` | create a task sharing the **caller's** address space |
|
||||
| `thread_exit` | `(stack_base, stack_len)` | end the calling thread; hand back its stack range for reclaim |
|
||||
| `futex_wait` | `(addr, expected, timeout_ns) -> status` | block if `*addr == expected`, until woken or timeout |
|
||||
| `futex_wake` | `(addr, count) -> woken` | wake up to `count` waiters on `addr` |
|
||||
|
||||
Plus one invariant change with no new syscall: **address-space reference counting**.
|
||||
|
||||
## Mechanics
|
||||
|
||||
### Address-space reference counting
|
||||
|
||||
Today an address space is 1:1 with a task: `spawnUserLocked` records `address_space` on the
|
||||
Task, and teardown does `destroyAddressSpace(t.address_space)` when **any** user task exits
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `address_space`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root (`createAddressSpace` in
|
||||
[process.zig](../system/kernel/process.zig) sets it to 1). `thread_spawn` increments
|
||||
it; task teardown decrements and only calls `destroyAddressSpace` at **zero**. All of
|
||||
this is already under the big kernel lock, so no new locking. This is the one piece
|
||||
that must land and be proven before anything shares an address space.
|
||||
|
||||
### `thread_spawn` and the trampoline
|
||||
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle values
|
||||
through registers — `startUserTask` reads the entry/stack from the Task and
|
||||
`jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread
|
||||
path clean:
|
||||
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`), heap-allocates a closure —
|
||||
`{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the
|
||||
closure pointer to the **top word of the new stack**.
|
||||
2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`.
|
||||
The kernel calls the same `spawnUserLocked` path with the **caller's address space**
|
||||
(refcount++), `entry`, and `user_sp = stack_top`.
|
||||
3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls
|
||||
the user function, then calls `thread_exit`. No new register ABI — the closure
|
||||
pointer rides the stack the runtime set up, mirroring how `startUserTask` avoids
|
||||
register smuggling.
|
||||
|
||||
Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||
([sysv.md](sysv.md)) — a thread stack carries only the closure pointer.
|
||||
|
||||
### Lifetime: exit, join, detach, stack reclaim
|
||||
|
||||
- **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack
|
||||
range. The kernel reaps the task on the scheduler (already running on a *kernel*
|
||||
stack, so it can safely unmap the user stack), decrements the address-space refcount, and
|
||||
frees the task slot.
|
||||
- **`join` — Stage 1** reuses the existing exit-notification machinery
|
||||
([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread
|
||||
`exit_endpoint`, and `join` blocks in `ipc_reply_wait` until the child-exit
|
||||
notification for that `tid` arrives, then `munmap`s the stack. No futex needed to
|
||||
land spawn/join.
|
||||
- **`join` — Stage 2 refinement** migrates to the std shape: a `completion` word in
|
||||
the closure that `thread_exit`'s trampoline `futex_wake`s and `join` `futex_wait`s
|
||||
on — dropping the per-thread endpoint. Kept as a refinement so Stage 1 ships first.
|
||||
- **`detach`** relinquishes the join right; the kernel reclaims the stack and slot on
|
||||
`thread_exit` (a detached thread's stack range is unmapped by the reaper, since no
|
||||
joiner will).
|
||||
|
||||
### Futex, and the sync primitives on top
|
||||
|
||||
`futex_wait`/`futex_wake` are the one blocking primitive; `Mutex`, `Condition`, and
|
||||
`Semaphore` are ordinary user-space state machines over an `atomic.Value(u32)` that
|
||||
call the futex wrappers on the slow path — the same construction `std.Thread` uses,
|
||||
so the algorithms port directly.
|
||||
|
||||
Keying: threads share an address space, so a **virtual address within that address space**
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shm](display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
other work or `hlt` ([halting.md](halting.md)). This is why futex is a locked
|
||||
decision, not a "maybe later."
|
||||
|
||||
### TLS and `getCurrentId`
|
||||
|
||||
danos sets up no thread-pointer TLS today (fine under `single_threaded`). Two scoped needs:
|
||||
|
||||
- **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better,
|
||||
a value the runtime stashes in a per-thread control block.
|
||||
- **`threadlocal` variables** need a real per-thread TLS block and the thread pointer set per
|
||||
thread. `thread_spawn` sets the thread pointer to a runtime-allocated per-thread block; full
|
||||
`threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core
|
||||
spawn/join/mutex path requires `threadlocal`.
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
`addUserBinary` gains a `threaded: bool = false` parameter; when set it builds that
|
||||
binary `single_threaded = false` so atomics and (later) TLS are real. Threads and
|
||||
atomics are unsound in a `single_threaded` image, so a binary must opt in **before**
|
||||
it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean.
|
||||
|
||||
## Interaction with the rest of the kernel
|
||||
|
||||
- **Scheduler / SMP** ([scheduling.md](scheduling.md), [smp.md](smp.md)): a thread is
|
||||
just another `Task` with an `address_space` shared with its siblings; the existing
|
||||
per-core ready queues, priorities, and affinity apply unchanged. Threads of one
|
||||
process can run on different cores simultaneously — that is the point.
|
||||
- **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core
|
||||
halts" property intact under lock contention — no busy-wait.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process
|
||||
must kill *all* its threads and only then drop the last address-space ref. The kill path
|
||||
already targets a process; it fans out to every task on that address space.
|
||||
- **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole
|
||||
process (shared fate). The supervisor restarts the **process**, which respawns its
|
||||
threads from a known-good state — restart granularity stays the process.
|
||||
- **IPC — two consequences threads forced ([ipc.md](ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||
to poke it awake (docs/display.md).
|
||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||
`send` already did — the kernel heap has no lock of its own yet (heap.zig: "a lock comes
|
||||
with threads/SMP"), so the big lock is what keeps its callers serialized.
|
||||
|
||||
## Build-out plan (staged, each gate serial-checkable)
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
- **Stage 0 — address-space refcount.** Refcount on the address-space root; teardown destroys
|
||||
at zero. No API yet; nothing shares an address space, so refcount is 1 everywhere.
|
||||
*Gate:* the full QEMU suite stays green (no regression) — proves the reframing is
|
||||
invisible until used.
|
||||
- **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline,
|
||||
stacks via `mmap`, join over the exit-endpoint, the `threaded` build flag.
|
||||
*Gate:* `-Dtest-case=thread-spawn` — a threaded test service spawns N threads that
|
||||
each `@atomicRmw`-increment a shared counter, the parent joins all N, and asserts
|
||||
the total is exactly N × iterations. Runs `smp` (multi-core) to prove real
|
||||
parallelism.
|
||||
- **Stage 2 — blocking synchronization.** `futex_wait`/`futex_wake` + `Futex`,
|
||||
`Mutex`, `Condition`, `Semaphore`; optionally migrate join to a futex completion
|
||||
word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a
|
||||
`Mutex` + `Condition` moves K items with no lost wakeups and no busy-wait (assert
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||
are — user code never names a syscall.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- **No preemptive user-space signals delivered to a specific thread.** Signals stay
|
||||
process-scoped ([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **No thread priorities distinct from the process.** Threads inherit the process
|
||||
priority; per-thread priority is a later question if it ever earns its keep.
|
||||
- **No cross-process shared-memory futex yet** — the physical-address key leaves the
|
||||
door open, but the first cut is private-per-address-space.
|
||||
- **No `pthread`/POSIX surface.** The API is `std.Thread`-shaped Zig, nothing more.
|
||||
|
||||
## The self-hosting endgame
|
||||
|
||||
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||
implementation, not a single call site. Designing to the std shape now is what makes
|
||||
the later self-hosting lift cheap.
|
||||
|
||||
## Further reading
|
||||
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||
+117
@@ -0,0 +1,117 @@
|
||||
# Timers and time
|
||||
|
||||
Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
- **Reading the clock** — *what time is it?* A read of a free-running counter.
|
||||
- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.*
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
|
||||
## Why time is a syscall, not a service
|
||||
|
||||
The tempting microkernel move is to put a timer *driver* in user space and have
|
||||
applications ask it for the time over IPC. For a **monotonic clock that is wrong** —
|
||||
reading `now()` should never cost an IPC round trip. The kernel is already holding the
|
||||
answer: it computes the current time every time it schedules, from the TSC, in a couple
|
||||
of instructions. Surfacing that as a system call is pure mechanism; routing it through a
|
||||
message to another process would be slower *and* redundant, and a device like the HPET
|
||||
(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`.
|
||||
|
||||
This is the same conclusion every serious system reaches: Linux and Zircon read the
|
||||
counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the
|
||||
cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall.
|
||||
|
||||
That "from the TSC" hides a portability question, because the TSC is only a valid clock
|
||||
when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*.
|
||||
danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
## The three system calls
|
||||
|
||||
Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)):
|
||||
|
||||
- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not
|
||||
wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a
|
||||
128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of
|
||||
resolution, and just an `rdtsc` plus a multiply.
|
||||
- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake
|
||||
deadline and the tick sweep wakes it (`scheduler.sleep`).
|
||||
- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a
|
||||
**timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
both riding the scheduler tick.
|
||||
|
||||
## `runtime.time` — the generic interface
|
||||
|
||||
Applications don't call the syscalls directly; they use `runtime.time`
|
||||
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
front door, not new mechanism.
|
||||
|
||||
```zig
|
||||
const time = @import("runtime").time;
|
||||
|
||||
const start = time.now(); // Instant — monotonic
|
||||
doWork();
|
||||
const took = start.elapsed(); // Duration
|
||||
time.sleep(time.Duration.fromMillis(5)); // block ~5 ms
|
||||
|
||||
// A deadline delivered as a notification, so a service keeps serving meanwhile:
|
||||
_ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||
```
|
||||
|
||||
- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/
|
||||
fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's
|
||||
millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and
|
||||
returns early. All arithmetic saturates rather than wraps.
|
||||
- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` —
|
||||
built for deadline loops (`while (!deadline.reached()) …`).
|
||||
- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is
|
||||
calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller
|
||||
that needs real time can treat 0 as "unavailable" rather than assume it advances).
|
||||
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||
|
||||
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
|
||||
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||
owns covers every current use.
|
||||
|
||||
## Verifying it
|
||||
|
||||
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
|
||||
```
|
||||
$ zig build test # includes library/runtime/time.zig
|
||||
```
|
||||
|
||||
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||
`Duration`, read `now()` again, and the second reading is later — the kernel's timer
|
||||
driving a ring-3 program with no service in between.
|
||||
+206
@@ -0,0 +1,206 @@
|
||||
# The vDSO — the public system-call boundary
|
||||
|
||||
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||
> only supported way into the kernel — so the raw numbers can stay private,
|
||||
> be renumbered at will, and eventually be randomised per boot.
|
||||
|
||||
## Why: the ABI danos promises, and the one it doesn't
|
||||
|
||||
`system/abi.zig` is the **private** kernel ↔ runtime contract. Its header says
|
||||
so: the numbers are an implementation detail the runtime hides and may
|
||||
renumber, the same split as libSystem over the XNU syscalls on macOS or win32
|
||||
over the NT syscalls on Windows. Linux — with its world-visible, frozen
|
||||
syscall table — is the outlier, not the norm.
|
||||
|
||||
That stance has consequences the moment binaries exist that we don't rebuild
|
||||
ourselves:
|
||||
|
||||
1. **Third-party binaries** (docs/zig-self-hosting.md) must keep working across
|
||||
kernel updates. If they contain raw `syscall` instructions with today's
|
||||
numbers baked in, every renumbering breaks the world — the ABI would be
|
||||
*de facto* public no matter what the header says. Go on macOS made exactly
|
||||
this mistake: it issued XNU syscalls directly instead of going through
|
||||
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||
switched to the library like everyone else.
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||
module. The public boundary has to be expressible in the one calling
|
||||
convention every language speaks: the C ABI.
|
||||
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||
work if no user binary anywhere knows a number at build time. The binding
|
||||
must happen at *load time*, from something the kernel controls.
|
||||
|
||||
All three point at the same well-known shape: a **vDSO** (virtual dynamic
|
||||
shared object). The kernel carries a small blob of user-mode code, maps it
|
||||
into every process at spawn, and that blob — not the application — contains
|
||||
the `syscall` instructions. Fuchsia works exactly this way: its vDSO is the
|
||||
*only* kernel entry, version-matched by construction because the kernel itself
|
||||
injects it. Because the kernel and the blob ship as one artifact, there is
|
||||
**no version skew, no loader, no search path, and no shared file on disk** —
|
||||
which is what makes this the resilient way to have a private ABI
|
||||
(docs/resilience.md), where a conventional `ld.so` + `/lib/libdanos.so`
|
||||
arrangement would add a loader to every spawn and a single shared point of
|
||||
failure.
|
||||
|
||||
The public danos ABI then has exactly two layers, neither of which is
|
||||
`abi.zig`:
|
||||
|
||||
| Layer | Contract | Spoken by |
|
||||
|-------|----------|-----------|
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
|
||||
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||
per-language convenience, compiled into each binary from source, exactly as
|
||||
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||
*only* door.
|
||||
|
||||
## The blob
|
||||
|
||||
A single copy of the vDSO code lives in the kernel image (built by
|
||||
`build.zig` as a tiny freestanding object, embedded like the AP trampoline).
|
||||
At boot the kernel finalises it once — this is where randomised numbers would
|
||||
be patched in — and thereafter maps the **same physical pages** read-execute
|
||||
into every process's address space. The blob is:
|
||||
|
||||
- **Position-independent.** It is mapped at a per-process randomised base, so
|
||||
it must be PIC (rip-relative addressing only — no relocations to process).
|
||||
- **Stateless and re-entrant.** No writable data. Anything stateful belongs to
|
||||
the process, not the vDSO.
|
||||
- **Architecture-specific.** The x86-64 blob wraps `syscall`; an aarch64 blob
|
||||
wraps `svc #0`. It lives beside the other per-architecture kernel sources
|
||||
(`system/kernel/architecture/<arch>/`), selected the same way the
|
||||
`architecture` module is (docs/arch.md).
|
||||
|
||||
### Shape: a function table, not an ELF
|
||||
|
||||
A real `.so` with a dynamic symbol table is the conventional vDSO shape, but
|
||||
linking against one at load time needs a dynamic linker in every binary —
|
||||
machinery danos deliberately doesn't have. Instead the v1 shape is the
|
||||
simplest thing that is still a stable contract — a **function-pointer table**
|
||||
at the vDSO base:
|
||||
|
||||
```
|
||||
offset 0 u64 magic 'danosVDS' — a mapped-the-wrong-thing guard
|
||||
offset 8 u64 api_level incremented when the table grows
|
||||
offset 16 u64 count number of table entries that follow
|
||||
offset 24 u64 table[count] function pointers into the vDSO's own code
|
||||
```
|
||||
|
||||
Table *indices* are the public constants (published in a C header,
|
||||
`danos.h`), assigned once and append-only — the same discipline the IPC
|
||||
protocols use for operation values. The pointers point at stubs inside the
|
||||
blob; what those stubs put in `rax` is nobody's business but the kernel's.
|
||||
A language shim binds in one step: read the base from the init block, check
|
||||
the magic, keep the table pointer. Feature detection for a binary built
|
||||
against older headers is `count`/`api_level` — a kernel never removes or
|
||||
reorders entries.
|
||||
|
||||
(If danos ever grows a real dynamic linker, the same blob can additionally
|
||||
present an ELF `dynsym` without breaking the table — Fuchsia's vDSO is
|
||||
likewise both a mappable blob and a linkable `.so`. That is a later
|
||||
convenience, not a requirement.)
|
||||
|
||||
### Delivery: the auxiliary vector
|
||||
|
||||
The kernel already builds a System V entry block — argc, argv, envp
|
||||
terminator, **auxiliary vector** — on every new process's stack
|
||||
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||
address, and a language shim finds it the same portable way on every
|
||||
architecture.
|
||||
|
||||
## The function surface
|
||||
|
||||
One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
||||
`danos_`. The current `SystemCall` set maps directly; integer arguments and
|
||||
returns are `u64`, errors return as negative values exactly as today.
|
||||
|
||||
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
(virtual_address + physical_address), `msi_bind` (address + data), `shm_create` (virtual_address + handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost.
|
||||
|
||||
Grouped as `abi.zig` groups them:
|
||||
|
||||
| Group | Functions |
|
||||
|-------|-----------|
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shm_create`, `danos_shm_map`, `danos_shm_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write`, `danos_klog_read` |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
ids, `page_size`, the IPC message maximum — move to the public header too:
|
||||
they are wire values a Rust program needs verbatim. What stays private in
|
||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||
numbers and the trap convention.
|
||||
|
||||
## Enforcement, and an honest threat model
|
||||
|
||||
Renumbering only has teeth if the kernel **refuses syscalls that don't come
|
||||
from the vDSO**. The check is cheap: on kernel entry, the saved user `rip`
|
||||
must lie inside the calling process's vDSO mapping; otherwise the process is
|
||||
killed with a fault-class exit reason (its supervisor restarts or gives up,
|
||||
docs/process-lifecycle.md — a foreign-syscall attempt is a bug or an attack,
|
||||
never something to limp past). Fuchsia enforces exactly this.
|
||||
|
||||
What this buys, precisely:
|
||||
|
||||
- **ABI freedom** — the real prize. The numbers can change per release or per
|
||||
boot and nothing outside the kernel image cares. The private ABI stays
|
||||
actually private, permanently.
|
||||
- **A single audited chokepoint** for kernel entry, per process, at a
|
||||
randomised address.
|
||||
- **Raised bar for exploits**: shellcode can't issue a hard-coded `syscall`;
|
||||
it must first discover the per-process vDSO base (ASLR) and call through
|
||||
it.
|
||||
|
||||
What it does *not* buy: an attacker with arbitrary code execution in a
|
||||
process can still *call* the vDSO functions — they are mapped executable in
|
||||
that process, and return-oriented chains reach them. Syscall randomisation is
|
||||
hardening, not a security boundary; the security boundary remains the
|
||||
capability model (what the process's endpoints and device claims let it do).
|
||||
It is worth building anyway — for the ABI freedom first and the hardening
|
||||
second — but the design should never be sold as more than that.
|
||||
|
||||
## Migration
|
||||
|
||||
Phased so every step ships alone (the M-milestone discipline):
|
||||
|
||||
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||
tree keeps booting during the transition.
|
||||
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
kernel-internal). The QEMU suite passing proves the table carries the
|
||||
whole system.
|
||||
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||
randomisation patched into the blob at kernel init. A test boots with
|
||||
randomisation on and runs the full suite.
|
||||
4. **The other languages.** Publish `danos.h`; a Rust `danos-sys` crate wraps
|
||||
the table. This is also the seam `std.os.danos` calls through when the Zig
|
||||
self-hosting fork lands (docs/zig-self-hosting.md) — the vDSO is what
|
||||
makes that seam stable across kernel versions.
|
||||
|
||||
## What deliberately stays out
|
||||
|
||||
- **No dynamic linker, no `/lib/*.so`.** The vDSO is kernel-injected precisely
|
||||
so danos binaries can stay fully static above it. Sharing *library code*
|
||||
across processes stays what it is today: a service behind IPC, or source
|
||||
compiled into each binary.
|
||||
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md); files are
|
||||
still the VFS server's business over IPC.
|
||||
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
|
||||
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
|
||||
day read the calibrated TSC in user mode the same way — the blob is where
|
||||
such an optimisation would live — but that is an optimisation, not part of
|
||||
this design's contract.
|
||||
@@ -0,0 +1,170 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> VFS server (`system/services/vfs`), with mounted backends (the FAT server)
|
||||
> speaking the same protocol behind the router. The Zig source of truth is
|
||||
> `system/services/vfs/protocol.zig` (the `vfs-protocol` module), whose unit
|
||||
> tests pin the sizes and values below. This page is the **language-neutral
|
||||
> wire specification** of that contract — what a Rust or C client implements
|
||||
> ([vdso.md](vdso.md) explains why the IPC protocols, not the syscall
|
||||
> numbers, are danos's public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint is found by well-known service id
|
||||
(`ipc_lookup`, service id **1** = vfs).
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
|
||||
bytes. There is no multi-message request: paths and single reads/writes
|
||||
must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
|
||||
read bytes, a `FileStatus`, or a `DirectoryEntry`.
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel never parses any of this — it only moves the bytes
|
||||
(docs/syscall.md); files are entirely a user-space affair.
|
||||
|
||||
## Request header — 32 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `operation` | an **Operation** value (below) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
|
||||
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
|
||||
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
|
||||
| 28 | 4 | `flags` | open flags (below); else 0 |
|
||||
|
||||
## Reply header — 24 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
|
||||
| 16 | 4 | `len` | reply payload length in bytes |
|
||||
| 20 | 4 | — | padding |
|
||||
|
||||
On failure the router replies `status = -1`; a mounted backend's negative
|
||||
status is forwarded to the client verbatim. A richer errno vocabulary is
|
||||
future work — clients must treat *any* negative status as failure, not match
|
||||
on -1.
|
||||
|
||||
## Operations
|
||||
|
||||
Values are append-only and never renumbered (the same evolution rule every
|
||||
danos protocol follows); an unrecognised operation gets a `status = -1`
|
||||
reply.
|
||||
|
||||
| value | operation | request payload | reply |
|
||||
|------:|-----------|-----------------|-------|
|
||||
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
|
||||
| 1 | `close` | — (`node` set) | status only |
|
||||
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
|
||||
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
|
||||
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
|
||||
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
|
||||
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
|
||||
| 7 | `unmount` | the mount-point path | status only |
|
||||
| 8 | `mkdir` | the path | status only |
|
||||
| 9 | `unlink` | the path | status only |
|
||||
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — paths are absolute (`/mnt/usb/notes.txt`) or bare names
|
||||
(`greeting`); bare names resolve in the VFS's flat ramfs, absolute paths
|
||||
route through the mount table (below). The returned `node` is an id in the
|
||||
*router's* open table; clients never see a backend's own ids.
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||
shorter than requested is progress, not an error; a `len` of 0 means no
|
||||
forward progress — stop rather than spin.
|
||||
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount** — the one operation that passes a **capability**: the caller
|
||||
(a filesystem server, e.g. FAT) sends its own request endpoint as the
|
||||
`ipc_call` capability argument, and the router forwards everything under
|
||||
the mount point to it — speaking this same protocol, with paths rewritten
|
||||
relative to the mount. Prefixes match at path boundaries only
|
||||
(`/mnt/usb` never captures `/mnt/usbextra`); the longest matching prefix
|
||||
wins.
|
||||
- **rename** — same-directory rename only (the router requires old and new to
|
||||
resolve under one mount).
|
||||
|
||||
## Open flags
|
||||
|
||||
Bitwise OR in `Request.flags`, meaningful for `open` only:
|
||||
|
||||
| bit | name | meaning |
|
||||
|----:|------|---------|
|
||||
| 1 | `create` | create the file if it does not exist |
|
||||
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||
|
||||
## FileStatus — 24 bytes (the `status` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 8 | `size` | file size in bytes |
|
||||
| 8 | 4 | `kind` | a **NodeKind** value |
|
||||
| 12 | 4 | — | padding |
|
||||
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||
| 4 | 4 | `name_len` | length of the name that follows |
|
||||
| 8 | 8 | `size` | the entry's size in bytes |
|
||||
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||
|
||||
## NodeKind
|
||||
|
||||
Aligned to the FSH file-type table
|
||||
(docs/danos-file-system-hierarchy-FSH.md):
|
||||
|
||||
| value | kind |
|
||||
|------:|------|
|
||||
| 0 | regular file |
|
||||
| 1 | directory |
|
||||
| 2 | character device |
|
||||
| 3 | block device |
|
||||
| 4 | symbolic link |
|
||||
| 5 | fifo |
|
||||
| 6 | socket |
|
||||
|
||||
Clients should map unknown values to *regular* rather than reject — the
|
||||
table can grow.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
Open-node ids live in the server. A client that dies without closing leaks
|
||||
nothing permanently: the VFS subscribes to the kernel's published process-exit
|
||||
events (docs/process-lifecycle.md) and releases a dead client's handles,
|
||||
closing forwarded backend nodes best-effort. Ids are plain integers, not
|
||||
capabilities — the VFS trusts its callers with each other's ids today, which
|
||||
is acceptable while every client is part of the system image and worth
|
||||
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
|
||||
## Evolution rules
|
||||
|
||||
What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped** — the unit tests in `protocol.zig`
|
||||
pin them exactly so a refactor can't silently move them.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
|
||||
exactly 0.
|
||||
+13
-4
@@ -81,11 +81,20 @@ prerequisites.
|
||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md).
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||
own page tables — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
|
||||
ring 3, per-process page tables). The substrate everything else needs. *Next, and a
|
||||
prerequisite for the resilience and driver tracks.*
|
||||
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||
higher-half kernel with a physmap (the low half is user space), per-process
|
||||
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||
and TLB shootdown once a process has more than one thread.*
|
||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||
See [resilience.md](resilience.md).
|
||||
|
||||
@@ -0,0 +1,349 @@
|
||||
# Running Zig on danos: the self-hosting roadmap
|
||||
|
||||
A design note (not built yet) on the path to making danos a **real Zig target** — a
|
||||
target you can name (`-target x86_64-danos`) and, eventually, run the Zig compiler
|
||||
itself on. It is forward-looking, like [vision.md](vision.md): it sets a direction
|
||||
and the decisions that follow from it, so the code we write now bends toward it
|
||||
instead of away.
|
||||
|
||||
This note deliberately does **not** cover a text editor or terminal. Those are
|
||||
easier (single-process, I/O-bound) and fall out of the early phases here almost for
|
||||
free; the hard, shaping problem is the standard-library surface, so that is what
|
||||
this roadmap is about.
|
||||
|
||||
The analysis behind it was done against **Zig 0.16** (the pinned toolchain). Zig's
|
||||
standard library moves between releases — especially the parts described here — so
|
||||
treat upstream references as "the shape in 0.16.x," and expect to re-check them on a
|
||||
toolchain bump.
|
||||
|
||||
## The win condition
|
||||
|
||||
danos runs the Zig compiler when a bare
|
||||
|
||||
```
|
||||
zig build-exe hello.zig
|
||||
```
|
||||
|
||||
completes **on danos** and produces a runnable danos binary. Note the milestone is
|
||||
`build-exe`, not `zig build`: the `zig build` runner spawns child processes (the
|
||||
build steps), which needs a whole process-control surface danos does not have yet.
|
||||
A single `build-exe` needs none of that (see Phase 3). Reaching `build-exe` is
|
||||
"self-hosting"; reaching `zig build` is a later, separate lift.
|
||||
|
||||
### Non-goals
|
||||
|
||||
- **No Linux syscall/ABI emulation.** danos will not implement the Linux `syscall`
|
||||
interface so that stock `x86_64-linux` binaries run. That is a permanent
|
||||
compatibility treadmill and it inverts the microkernel design — explicitly out.
|
||||
- **No musl port yet.** A musl libc port is a reasonable *later* effort (it unlocks
|
||||
the C ecosystem), but it is not on the critical path to Zig-on-danos, and it is
|
||||
deferred. The roadmap below is arranged so the work still pays off if musl ever
|
||||
happens (see "The same surface, twice").
|
||||
- **Editor/terminal are out of scope for this note** (they are downstream of Phase 1).
|
||||
|
||||
**On FFI.** Foreign-function interop splits the same way as the doors below. Zig-level
|
||||
and C-ABI-*exposing* FFI (`extern`, `callconv(.c)`, C-ABI structs) work on a real target
|
||||
immediately — and the `std.os.danos` seam is C-ABI-shaped by construction, so it is
|
||||
FFI-friendly from the start. *Consuming* C libraries (`@cImport`, linking archives) is
|
||||
the part that needs a libc + headers, i.e. the deferred musl door. So an eventual FFI
|
||||
need reinforces keeping that door open; it does not change the plan.
|
||||
|
||||
## The realization that shapes everything: 0.16 gives us *one* seam
|
||||
|
||||
The instinct "to target Zig we'd have to reimplement all the `std` namespaces" was
|
||||
how older Zig worked. Zig 0.16 (post-"writergate") is far kinder:
|
||||
|
||||
- **`std.fs` is essentially gone.** It is now path helpers plus deprecated aliases;
|
||||
there is no `std.fs.File`, `std.fs.Dir`, or `std.fs.cwd()`. File and directory
|
||||
work goes through **`std.Io`** — a single runtime **vtable** (`Io.zig`) of
|
||||
function pointers handed to `main` as `std.process.Init.io`. `std.Io.File` and
|
||||
`std.Io.Dir` are thin forwarders to that vtable. `Io.zig` and the `fs` shim carry
|
||||
**zero** per-OS branches.
|
||||
- **`std.posix` is one generic body** parameterised over a single `system` module.
|
||||
With no libc, `system` resolves **per target OS**: `.linux => std.os.linux`,
|
||||
`.plan9 => std.os.plan9`, and so on. The generic `std.posix.read`/`write`/`open`
|
||||
bodies are just `system.read(...)` plus an errno switch — *identical for every
|
||||
OS*. The only variable is what `system` binds to.
|
||||
- **`std.os.<tag>`** (e.g. `std/os/linux.zig`) is therefore the real porting seam: a
|
||||
low-level, C-ABI-shaped module of `read/write/open/close/lseek/mmap/clock/exit/…`
|
||||
plus an `errno` enum and the constant tables (`O_*`, `CLOCK_*`, `S_*`).
|
||||
|
||||
Put together: **to port danos we write `std.os.danos` once** — the ~30-operation
|
||||
seam — and the whole `std.posix` / `std.fs` / `std.Io` tower above it lights up
|
||||
generically, because none of it branches on the OS. That is a dramatically smaller
|
||||
and more contained target than "reimplement the namespaces."
|
||||
|
||||
## Three doors, and why we take the first
|
||||
|
||||
| Door | What it is | Verdict |
|
||||
|------|-----------|---------|
|
||||
| **1. Implement the std seam** (`std.os.danos`) | Write the ~30-op `system` module over danos's native ABI + VFS; the generic std tower lights up. | **Take this.** The only door that touches neither C nor the Linux ABI. |
|
||||
| **2. Port musl** | Port musl libc to danos, link Zig against it. | Defer. Good later for the *C* ecosystem; barely helps *Zig* (std only uses libc on the libc-linked path). |
|
||||
| **3. Emulate the Linux ABI** | Implement Linux syscalls so stock linux binaries run. | Reject. Bottomless compatibility treadmill; against the design. |
|
||||
|
||||
### The same surface, twice
|
||||
|
||||
Doors 1 and 2 are the **same native surface at different layers**. `std.posix.read`
|
||||
is `system.read(...)` + an errno switch *regardless of OS* — the only question is
|
||||
whether `system` is **`std.os.danos` (Zig)** or **musl (C)**. Either way, the set of
|
||||
danos-facing operations you must implement is the *same* ~30 ops, all bottoming out
|
||||
in danos's native syscalls + the VFS/FAT server.
|
||||
|
||||
So the runtime work below is **not throwaway** if musl ever happens: you are building
|
||||
the danos-native implementations of that surface either way. Door 1 just packages
|
||||
them as Zig; a future musl re-uses the identical kernel/VFS operations underneath. The
|
||||
two symmetries worth keeping in mind: doors 1 and 2 converge at the **top** (identical
|
||||
POSIX surface); doors 2 and 3 converge at the **bottom** (unmodified musl needs the
|
||||
Linux syscall ABI). Door 1 is the only one that avoids both C and Linux.
|
||||
|
||||
### A fork is table stakes — for any door
|
||||
|
||||
`std.Target.Os.Tag` is a **closed enum** baked into the compiler binary *and* into
|
||||
the `std` linked with every program; `-target x86_64-danos` resolves through it. So
|
||||
adding `danos` as a name requires patching and rebuilding the compiler — even the
|
||||
musl door needs this. "Fork Zig" is therefore not an extra cost unique to door 1; it
|
||||
is the price of admission for *any* real target. What door 1 adds on top is small and
|
||||
localised (below).
|
||||
|
||||
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||
|
||||
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||
|
||||
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||
(`read/write/open/close/lseek/mmap/munmap/clock/exit/…`) + an errno enum + the
|
||||
constant tables, each backed by danos's native syscalls and the VFS. **Structure it
|
||||
to mirror `std/os/linux.zig`.** This is the load-bearing, *non-throwaway* artifact:
|
||||
when we fork Zig, `runtime.os` is copy-pasted (near-verbatim) into `std.os.danos`.
|
||||
- **`runtime.fs` — the thin native file API** danos programs use *today*, layered
|
||||
over `runtime.os`. It is also the concrete backing for the `std.Io` vtable's
|
||||
file-write entry once we're a real target, which is why program stdout, diagnostics,
|
||||
and file writes should all be *decided once at that seam* rather than as bespoke
|
||||
per-call helpers (see "How this informs decisions now").
|
||||
|
||||
**Do not hand-mirror the high-level std namespaces.** `std.fs`/`std.Io`/`std.process`
|
||||
are generic and OS-agnostic; once `std.os.danos` exists and we fork, upstream *gives*
|
||||
them to danos for free. Hand-writing `runtime.std.fs` to imitate them would be
|
||||
redundant the day the fork works, and it would chase a moving target (0.16's `std.Io`
|
||||
is large and still shifting). Build the seam well; take the tower for free.
|
||||
|
||||
**Why not a library called `std`?** Because `@import("std")` resolves to the
|
||||
compiler-provided standard library; a user module named `std` would *shadow* it for
|
||||
anything that imports it that way. That is the real reason the seam lives *inside* a
|
||||
forked std as `std/os/danos.zig`, not as a `runtime.std` library — and why danos's end
|
||||
state (`@import("std")` just working, and knowing danos) is the most natively Zig it can
|
||||
be. `runtime.os` is only the interim staging ground: developed against the stock
|
||||
toolchain so Phase 1 need not wait on the fork, then promoted near-verbatim into the
|
||||
fork's `std/os/danos.zig`.
|
||||
|
||||
### Retire `library/posix`
|
||||
|
||||
The `posix` compatibility layer (`unistd`, `stdio`) was the right instinct too early.
|
||||
Its whole value is POSIX *spellings* for POSIX software — and danos has no POSIX
|
||||
software; every current caller is danos-native code that could use `runtime.fs`
|
||||
directly. The real POSIX story arrives later and from elsewhere (musl, or upstream
|
||||
`std`'s own posix over `std.os.danos`), which supersedes a hand-rolled shim. So it is
|
||||
premature abstraction that adds a "which layer do I use?" fork with no payoff yet.
|
||||
|
||||
Its footprint is tiny: **five** call sites, all `unistd` file operations —
|
||||
`system/services/fat/fat.zig` (`mount`), the `vfs-test` and `fat-test` clients, and
|
||||
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` is dead — nothing
|
||||
imports it. The plan: build `runtime.fs`, migrate those five to it, delete
|
||||
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary`.
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
What the seam needs, and what danos already provides:
|
||||
|
||||
| std need | danos today | Gap |
|
||||
|----------|-------------|-----|
|
||||
| open / read / write / close / lseek | VFS (via the current `unistd`, → `runtime.fs`) | none — repackage |
|
||||
| directory read (`getdents`) | VFS `readdir` | none — repackage |
|
||||
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||
| monotonic clock | `clock` syscall | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||
| wall-clock / realtime | done — `wall_clock` syscall (CMOS RTC, Phase 2d) | — |
|
||||
| **environment variables** | `Init` has no env field | missing (can start empty) |
|
||||
| **cwd / chdir** | paths are absolute or bare | missing (no cwd anchor) |
|
||||
| **entropy / random** | — | missing (needed behind `vtable.random`) |
|
||||
| process spawn + exit status | `system_spawn` starts a *named ramdisk binary*; `ExitReason` is a *category* | no exec-of-path, no numeric `WEXITSTATUS` |
|
||||
| threads | one thread per process | avoided via `-fsingle-threaded` (below) |
|
||||
| symlinks | `NodeKind` has the tag; unimplemented | low priority |
|
||||
|
||||
The clustering is clear: reads and memory are basically done; the real work is
|
||||
**filesystem mutation + richer stat + wall-clock**, and a few small seam pieces
|
||||
(page-allocator hook, stdio bytes, entropy). Process spawning and threads are
|
||||
side-stepped entirely for a single `build-exe`.
|
||||
|
||||
## The roadmap
|
||||
|
||||
### Phase 0 — Make `danos` a real target
|
||||
|
||||
**Host, target, self-host — keep the three roles straight.** The *host* is where the
|
||||
compiler runs (your mac + linux dev machines); the *target* is what it emits (`danos`);
|
||||
and eventually danos becomes a host too (self-hosting — the win condition). So the move
|
||||
is: fork the compiler, build it **for** your dev hosts, and teach it to **cross-compile
|
||||
to** danos. You already do this — danos is cross-compiled `freestanding` from your dev
|
||||
host today; Phase 0 swaps that `freestanding` target for a real `x86_64-danos` one, which
|
||||
is what unlocks the native `std`.
|
||||
|
||||
**Why a compiler fork, not just a `--zig-lib-dir` override.** `std.Target.Os.Tag` is a
|
||||
*closed enum compiled into the compiler binary*, so `-target x86_64-danos` will not even
|
||||
parse unless the compiler itself knows the tag. Overriding the std lib directory alone
|
||||
cannot add a target — and there is no libc-only shortcut (a future musl needs the same
|
||||
patch). The only alternative, staying on `freestanding` + hand-shims, is exactly the
|
||||
non-native feel we are leaving: `@import("std")` there is stubbed, not real.
|
||||
|
||||
**The fork.** Clone `ziglang/zig` at the pinned 0.16 tag; build it with a stock
|
||||
same-version `zig` (`zig build` in the tree — a standard, LLVM-pulling, roughly one-time
|
||||
build); point danos's `build.zig`/CI at the resulting binary. Four localised patches:
|
||||
|
||||
- add `danos` to `std.Target.Os.Tag`, in the "no version range" group alongside
|
||||
plan9/serenity;
|
||||
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||
not a prerequisite for starting).
|
||||
|
||||
This is the fork treadmill we accept once. Keep the patch set tiny and `else`-friendly,
|
||||
pin to one 0.16.x, and rebase on point releases.
|
||||
|
||||
### Phase 1 — `runtime.os` read-side + allocator + stdio + cwd; retire `posix`
|
||||
|
||||
Author `runtime.os` (→ `std.os.danos`): the `errno` enum, the constant tables, and
|
||||
the C-convention `read / write / open / openat / close / lseek / mmap / munmap /
|
||||
exit`, each returning result-or-`-errno`. Most backing already exists (VFS + native
|
||||
mmap + clock).
|
||||
|
||||
- Provide `page_allocator` via `root.os.heap.page_allocator` (a thin override over
|
||||
danos `mmap`). This sits **outside** the `std.Io` vtable, so it is wired separately.
|
||||
- Wire fd 0/1/2 to a console **byte** stream (today output only reaches `debug_write`;
|
||||
input is structured `InputEvent` IPC — a byte tty is a new, small thing in both
|
||||
directions).
|
||||
- Add a `getcwd`/`chdir` anchor so `std.fs.cwd()`-style resolution has something to
|
||||
resolve against.
|
||||
- Build `runtime.fs` over `runtime.os`; migrate the five `posix` callers to it; delete
|
||||
`library/posix/` and drop its build module.
|
||||
|
||||
After Phase 1, the surface an editor or terminal needs (open/read/write/close/lseek/
|
||||
readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||
|
||||
### Phase 2 — Filesystem mutation + real stat (the compiler's cache tower)
|
||||
|
||||
danos's biggest genuine gap, and the correctness-critical one:
|
||||
|
||||
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||
([protocol.zig](../system/services/vfs/protocol.zig)) and the FAT engine
|
||||
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||
(danos is monotonic-only today; an RTC/time service is the dependency).
|
||||
|
||||
Because `std.fs`/`std.Io` have no per-OS branches, finishing this in `runtime.os`
|
||||
lights up the whole file tower for the compiler at once. Environment can stay an empty
|
||||
map until the kernel populates a non-empty `envp`.
|
||||
|
||||
**Status — Phase 2 complete.** `truncate` (O_TRUNC, closing the boot-log stale-tail
|
||||
bug), `mkdir`, `unlink`, and `rename` are all wired through the FAT engine, the VFS
|
||||
protocol + router, and `runtime.fs` (`makeDirectory` / `remove` / `rename`) —
|
||||
host-tested and QEMU-tested (`fat-mutations` + `fat-rename` make a directory, write+read
|
||||
a file in it, rename it, then remove it through the mount). `removeFile` and `rename`
|
||||
are LFN-aware; `rename` is same-directory + 8.3 (cross-directory and long-name-
|
||||
preserving rename are noted limitations). Wall-clock is now a kernel syscall
|
||||
(`wall_clock`, a CMOS-RTC read anchored to the monotonic clock), and the FAT engine
|
||||
stamps and reports **mtime** — `stat` / `runtime.fs.Attributes` carry a real
|
||||
modification time (the `fat-mtime` case reads it back within seconds of the host clock).
|
||||
The remaining `stat` fields, `mode`/`inode`, are deferred (not needed until the
|
||||
compiler's cache layer wants them). **Everything past here is gated on Phase 0 (the
|
||||
fork):** the `runtime.os` seam, `cwd`, stdio-as-fds, and the compiler bring-up.
|
||||
|
||||
### Phase 3 — Single-threaded, self-linked compiler bring-up
|
||||
|
||||
Build the compiler with **two load-bearing flags**:
|
||||
|
||||
- **`-fsingle-threaded`** removes `std.Thread` entirely — `Thread.spawn` is a hard
|
||||
compile error under it, and `std.Io`'s threaded backend runs inline. danos being
|
||||
one-thread-per-process is therefore **not** a blocker. Parallel codegen is a
|
||||
throughput optimisation, not a correctness requirement.
|
||||
- **`-fno-llvm -fno-lld`** keeps codegen and linking **in-process** (the self-hosted
|
||||
x86-64 backend + self-linker), so a single `build-exe` **never forks a child**. That
|
||||
is what lets us defer the entire spawn/exec/wait surface.
|
||||
|
||||
Then supply the few remaining seam pieces: `now` (wrap the danos clock), an entropy
|
||||
source behind `vtable.random` (`randomSecure` can alias it initially — low volume, for
|
||||
temp-file names and hashmap seeds), and the Phase-2 mkdir/rename/unlink for cache dir
|
||||
trees and atomic temp-then-rename output.
|
||||
|
||||
**Explicitly deferred** (not on the `build-exe` path): child-process spawn/exec (only
|
||||
`zig build` and external tools need it), `std.Thread`, `fsync` (FAT is write-through
|
||||
today), symlinks, and musl.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **The std-fork rebase treadmill is the main ongoing cost.** A new OS tag touches the
|
||||
same broad file set plan9/serenity touch (hundreds of `native_os` sites, plus
|
||||
"unsupported OS" `@compileError` dead-ends a new tag must be routed around), and the
|
||||
entire `std.Io` layer is new in 0.16 and still moving. Stay pinned to one 0.16.x,
|
||||
keep additions localised and `else`-friendly. Watch the closed-enum gotcha: adding
|
||||
`danos` to `Os.Tag` can break existing *exhaustive* switches that lack an `else`, so
|
||||
expect to touch switch sites beyond the ones you implement.
|
||||
- **Single-threaded is load-bearing.** The "no `std.Thread`" simplification rests
|
||||
entirely on `-fsingle-threaded`. If a dependency or flag flips threading back on, you
|
||||
inherit an unescapable compile error (no root-hook exists) — the only outs are a full
|
||||
thread-impl fork or linking libc for pthreads. Keep `single_threaded` asserted end to
|
||||
end.
|
||||
- **In-process linking is load-bearing.** Reaching the compiler without fork/exec
|
||||
depends on `-fno-llvm -fno-lld`. The moment you shell out to LLD/`ld`, you need the
|
||||
full `spawn`/`wait` surface — the hardest microkernel piece — and danos's
|
||||
`system_spawn` only starts a *named ramdisk binary*, not exec of an arbitrary path.
|
||||
Verify the self-hosted backend covers the target output before assuming child
|
||||
processes are optional.
|
||||
- **The shim cannot host the compiler.** danos's current `runtime`/`posix` is fine for
|
||||
danos's *own* native programs, but the compiler `import`s *upstream* `std`, which on
|
||||
a non-target hits the void `system` stub. So the compiler forces the real target
|
||||
(Phase 0's fork). Do not over-invest in extending the hand-shim for compiler
|
||||
purposes; put that effort into `runtime.os` + the VFS/FAT operations, which both the
|
||||
fork *and* a future musl consume.
|
||||
- **`"w"`/`O_CREAT` does not truncate — a silent-corruption bug on this road.** The FAT
|
||||
engine's `writeFile` only *grows* `node.size`, so overwriting a shorter file leaves
|
||||
trailing garbage. Harmless for the boot log today, but for a compiler it means
|
||||
**corrupt `.o`/cache files that look like nondeterministic compiler bugs.** Land
|
||||
`truncate` (Phase 2) before the compiler ever writes cache.
|
||||
- **Exit status is categorical, not numeric.** `process_exit_reason` returns an
|
||||
`ExitReason` *category*, not a numeric code (`WEXITSTATUS`). Fine while spawn is
|
||||
stubbed; the day `zig build` or external tools arrive, plan a kernel exit-record
|
||||
extension — do not let it surprise you.
|
||||
|
||||
## How this informs decisions now
|
||||
|
||||
Two current decisions fall out of this roadmap:
|
||||
|
||||
1. **The `runtime.fs` / `std.Io` question resolves at the vtable seam.** Because 0.16
|
||||
routes *all* output through the `std.Io` vtable's file-write entry, and stdout/stderr
|
||||
are just `File`s with well-known handles, build `runtime.fs` (and the console stdout)
|
||||
as the concrete backing for that entry — not as a bespoke `std.Io.Writer`-only shim.
|
||||
Decide it once, at the seam, and program stdout, diagnostics, and file writes all
|
||||
flow through the same danos VFS/console path.
|
||||
2. **The boot-log `truncate` caveat is now fixed** (Phase 2a). It was the same
|
||||
`writeFile`-only-grows gap that on the self-hosting road would corrupt build output;
|
||||
`engine.truncate` + an O_TRUNC open flag now free the old chain so a shorter rewrite
|
||||
leaves no stale tail, and the boot-log flush opens with it.
|
||||
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the north star this serves.
|
||||
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
are confined, and now retired).
|
||||
@@ -0,0 +1,80 @@
|
||||
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||
//!
|
||||
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||
//! don't reorder it against *other volatile* accesses. It says nothing about ordinary
|
||||
//! stores — the DMA descriptor you just filled in write-back RAM — which the compiler
|
||||
//! (and, on weakly-ordered hardware, the CPU) may freely move past a volatile MMIO
|
||||
//! write. The canonical bug:
|
||||
//!
|
||||
//! ring[i] = descriptor; // ordinary store to WB RAM
|
||||
//! doorbell.* = i; // volatile store to UC MMIO
|
||||
//! // nothing orders these; the device can read a stale descriptor
|
||||
//!
|
||||
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||
//!
|
||||
//! x86_64 aarch64
|
||||
//! mb() mfence dsb sy
|
||||
//! rmb() lfence dsb ld
|
||||
//! wmb() sfence dsb st
|
||||
//!
|
||||
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||
//! abstraction exists now, while there is one caller (hpet) to get right. See
|
||||
//! docs/driver-model.md (M14) for the full ordering contract.
|
||||
|
||||
const builtin = @import("builtin");
|
||||
|
||||
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||
/// another volatile access.
|
||||
pub inline fn read(comptime T: type, addr: usize) T {
|
||||
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||
}
|
||||
|
||||
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||
/// it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn mb() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.mb: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// wake, before reading what the device wrote to shared memory.
|
||||
pub inline fn rmb() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn wmb() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
test "barriers emit and registers round-trip through a RAM cell" {
|
||||
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||
// tested, but a missing/mistyped mnemonic is caught here.
|
||||
wmb();
|
||||
rmb();
|
||||
mb();
|
||||
var cell: u64 = 0;
|
||||
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
//! Block-device client: the helper a filesystem uses to read and write a block
|
||||
//! device (a USB stick, via usb-storage) without hand-rolling the block-protocol
|
||||
//! IPC. Layered over `ipc` and the shared `block-protocol` wire format, like
|
||||
//! `runtime.usb` over the transfer protocol.
|
||||
//!
|
||||
//! Transfers name a caller-owned DMA buffer by physical address (from
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("block-protocol");
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
endpoint: ipc.Handle,
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (result.status != 0) return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.write, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// Look up the block device, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 1200) : (attempts += 1) {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
pub const DeviceClass = device_abi.DeviceClass;
|
||||
pub const ResourceKind = device_abi.ResourceKind;
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||
pub fn claim(id: u64) bool {
|
||||
return !failed(sc.systemCall1(.device_claim, id));
|
||||
}
|
||||
|
||||
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||
/// `device_id` into this address space; returns the register base virtual address.
|
||||
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||
pub const no_parent = device_abi.no_parent;
|
||||
|
||||
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
||||
/// descriptors passed to `register` unless the child really is one.
|
||||
pub const no_pci_class = device_abi.no_pci_class;
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||
///
|
||||
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||
/// through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||
}
|
||||
|
||||
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||
/// status register holds its line asserted) — the kernel left the line masked
|
||||
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||
}
|
||||
|
||||
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
||||
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
||||
pub const Msi = struct { address: u64, data: u32 };
|
||||
|
||||
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
||||
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
||||
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
||||
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
||||
/// walking its capability list. Returns null on failure. Two return values (address in
|
||||
/// rax, data in rdx), so a hand-written stub.
|
||||
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined;
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
||||
[a0] "{rdi}" (device_id),
|
||||
[a1] "{rsi}" (endpoint),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
||||
/// null if the capability check fails (device not claimed, wrong resource, out of
|
||||
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
||||
/// a failure.
|
||||
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
||||
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
||||
return if (failed(r)) null else @intCast(r);
|
||||
}
|
||||
|
||||
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
||||
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
||||
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
||||
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
||||
}
|
||||
|
||||
/// Find DeviceDescription by hid
|
||||
///
|
||||
/// Utility function for driver development
|
||||
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
||||
const total = enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
||||
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
||||
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
||||
return d;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("display-protocol");
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
};
|
||||
|
||||
/// The service endpoint, looked up once and cached.
|
||||
var handle: ?ipc.Handle = null;
|
||||
|
||||
/// Look up the display service, retrying while it comes up (a client races its
|
||||
/// registration at boot). Returns the endpoint, or null if it never appears.
|
||||
fn service() ?ipc.Handle {
|
||||
if (handle) |h| return h;
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.display)) |h| {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||
/// so callers can read `info`/`layer` fields on success.
|
||||
fn transact(request: protocol.Request, out: *protocol.Reply) bool {
|
||||
const h = service() orelse return false;
|
||||
var req = request;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
return out.status == 0;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
var reply: protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(protocol.Operation.info) }, &reply)) return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
fn cachedInfo() ?Info {
|
||||
if (mode) |m| return m;
|
||||
const i = info() orelse return null;
|
||||
mode = i;
|
||||
return i;
|
||||
}
|
||||
|
||||
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||
pub const Layer = struct {
|
||||
id: u32,
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.fill_rect),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.colour = colour,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||
/// `protocol.maximum_payload`.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
var request = protocol.Request{
|
||||
.operation = @intFromEnum(protocol.Operation.blit_tile),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
};
|
||||
const header = std.mem.asBytes(&request);
|
||||
if (header.len + pixels.len > protocol.message_maximum) return false;
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(buffer[0..header.len], header);
|
||||
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||
const h_svc = service() orelse return false;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.configure_layer),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.z = z,
|
||||
.visible = if (visible) 1 else 0,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.damage),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
var reply: protocol.Reply = undefined;
|
||||
if (!transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.create_layer),
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.z = z,
|
||||
.visible = 1,
|
||||
}, &reply)) return null;
|
||||
return .{ .id = reply.layer };
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||
/// opt-in for specific hardware — see `abi`.
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
pub fn free(virtual: usize, len: usize) void {
|
||||
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||
}
|
||||
@@ -0,0 +1,274 @@
|
||||
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||
//! lists files served by the user-space VFS (system/services/vfs), each call
|
||||
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||
//! danos programs use directly; it is also where the file operations that later
|
||||
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||
//! the old POSIX `unistd` shim — a compatibility spelling danos does not need yet.
|
||||
//!
|
||||
//! Handles are *values*, not entries in a global descriptor table: a `File` /
|
||||
//! `Directory` owns its VFS node id and (for files) a byte offset. So there is no
|
||||
//! per-process fd limit and no shared table to synchronise — the danos-native
|
||||
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
/// wire protocol.
|
||||
pub const Kind = protocol.NodeKind;
|
||||
|
||||
/// A node's metadata (the answer to a status request).
|
||||
pub const Attributes = struct {
|
||||
size: u64,
|
||||
kind: Kind,
|
||||
/// Modification time — Unix epoch seconds, UTC. 0 if the filesystem has none.
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
// Map a wire `NodeKind` value to the enum, defaulting anything unrecognised to
|
||||
// `.regular` (the server is trusted, but a value outside the enum would be
|
||||
// illegal to `@enumFromInt` directly).
|
||||
fn kindFromWire(value: u32) Kind {
|
||||
return switch (value) {
|
||||
@intFromEnum(Kind.directory) => .directory,
|
||||
@intFromEnum(Kind.character_device) => .character_device,
|
||||
@intFromEnum(Kind.block_device) => .block_device,
|
||||
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||
@intFromEnum(Kind.fifo) => .fifo,
|
||||
@intFromEnum(Kind.socket) => .socket,
|
||||
else => .regular,
|
||||
};
|
||||
}
|
||||
|
||||
/// How to open a path.
|
||||
pub const OpenOptions = struct {
|
||||
/// Create the file if it does not exist.
|
||||
create: bool = false,
|
||||
/// Open a directory node (for listing) rather than a file.
|
||||
directory: bool = false,
|
||||
/// Truncate an existing file to zero length on open (O_TRUNC) — replace its
|
||||
/// contents rather than overwriting in place.
|
||||
truncate: bool = false,
|
||||
|
||||
fn wireFlags(self: OpenOptions) u32 {
|
||||
var f: u32 = 0;
|
||||
if (self.create) f |= protocol.create;
|
||||
if (self.directory) f |= protocol.directory;
|
||||
if (self.truncate) f |= protocol.truncate;
|
||||
return f;
|
||||
}
|
||||
};
|
||||
|
||||
// The VFS server endpoint, looked up once by well-known id and cached.
|
||||
var vfs_handle: ipc.Handle = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?ipc.Handle {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||
const rpl = @min(n - protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||
pub const File = struct {
|
||||
node: u64,
|
||||
offset: u64 = 0,
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
/// call is capped at the VFS payload size, so the return may be short — use
|
||||
/// `writeAll` to write the whole slice. Null on error.
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||
/// total written, or null if a write failed before any progress.
|
||||
pub fn writeAll(self: *File, data: []const u8) ?usize {
|
||||
var written: usize = 0;
|
||||
while (written < data.len) {
|
||||
const n = self.write(data[written..]) orelse return if (written == 0) null else written;
|
||||
if (n == 0) return written; // no forward progress; stop rather than spin
|
||||
written += n;
|
||||
}
|
||||
return written;
|
||||
}
|
||||
|
||||
/// Move the read/write cursor to an absolute byte position.
|
||||
pub fn seekTo(self: *File, position: u64) void {
|
||||
self.offset = position;
|
||||
}
|
||||
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this file.
|
||||
pub fn close(self: *File) void {
|
||||
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
|
||||
const r = transact(request, path, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node };
|
||||
}
|
||||
|
||||
/// A path's metadata without keeping it open (open -> status -> close).
|
||||
pub fn attributes(path: []const u8) ?Attributes {
|
||||
var file = open(path, .{}) orelse return null;
|
||||
defer file.close();
|
||||
return file.attributes();
|
||||
}
|
||||
|
||||
/// Whether `path` resolves — handy as a readiness check (e.g. waiting for a mount
|
||||
/// to come up before writing to it).
|
||||
pub fn exists(path: []const u8) bool {
|
||||
return attributes(path) != null;
|
||||
}
|
||||
|
||||
/// One entry returned by `Directory.next`.
|
||||
pub const Entry = struct {
|
||||
kind: Kind = .regular,
|
||||
size: u64 = 0,
|
||||
name_buffer: [64]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
|
||||
pub fn name(self: *const Entry) []const u8 {
|
||||
return self.name_buffer[0..self.name_len];
|
||||
}
|
||||
};
|
||||
|
||||
/// An open directory being listed, cursor-advanced by `next`.
|
||||
pub const Directory = struct {
|
||||
node: u64,
|
||||
cursor: u64 = 0,
|
||||
|
||||
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||
entry.kind = kindFromWire(header.kind);
|
||||
entry.size = header.size;
|
||||
const source = r.payload[protocol.directory_entry_size..];
|
||||
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this directory.
|
||||
pub fn close(self: *Directory) void {
|
||||
var f = File{ .node = self.node };
|
||||
f.close();
|
||||
}
|
||||
};
|
||||
|
||||
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||
pub fn openDirectory(path: []const u8) ?Directory {
|
||||
const file = open(path, .{ .directory = true }) orelse return null;
|
||||
return .{ .node = file.node };
|
||||
}
|
||||
|
||||
// A path-based request that returns only a status (mkdir, unlink).
|
||||
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
|
||||
const r = transact(request, path, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||
/// success. Only works under a mounted filesystem that supports directories.
|
||||
pub fn makeDirectory(path: []const u8) bool {
|
||||
return pathOperation(.mkdir, path);
|
||||
}
|
||||
|
||||
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||
/// (a separate directory-removal would have to check emptiness).
|
||||
pub fn remove(path: []const u8) bool {
|
||||
return pathOperation(.unlink, path);
|
||||
}
|
||||
|
||||
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
|
||||
/// directory, 8.3-name rename only for now). Returns true on success.
|
||||
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const total = old_path.len + 1 + new_path.len;
|
||||
if (total > protocol.maximum_payload) return false;
|
||||
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_path.len], old_path);
|
||||
payload[old_path.len] = 0;
|
||||
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
/// the VFS then routes everything under `target` to that backend. This is the one
|
||||
/// call that hands the VFS a capability (the backend endpoint). Returns true on
|
||||
/// success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
const h = vfs() orelse return false;
|
||||
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const tlen = @min(target.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
|
||||
if (result.len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
@@ -0,0 +1,215 @@
|
||||
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||
//! and Zig `std` containers share one heap.
|
||||
//!
|
||||
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||
//! that changes on this side of the system_call boundary is where memory comes from
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
|
||||
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
|
||||
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
|
||||
//! compiles it out and pays nothing, while a threaded one can allocate safely from
|
||||
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
|
||||
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
|
||||
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
|
||||
var heap_mutex: Mutex = .{};
|
||||
|
||||
inline fn lockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.lock();
|
||||
}
|
||||
inline fn unlockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.unlock();
|
||||
}
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||
const chunk = 64 * 1024;
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||
if (system_calls.mmapFailed(ret)) return false;
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
|
||||
/// across the free-list search and any `grow` (which also touches the free list).
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- C ABI: the global implicit heap ---------------------------------------
|
||||
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||
|
||||
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||
if (size == 0) return null;
|
||||
const p = rawAlloc(size) orelse return null;
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||
const p = ptr orelse return;
|
||||
rawFree(@ptrCast(p));
|
||||
}
|
||||
|
||||
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||
if (total == 0) return null;
|
||||
const p = rawAlloc(total) orelse return null;
|
||||
@memset(p[0..total], 0);
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||
const p = ptr orelse return malloc(size);
|
||||
if (size == 0) {
|
||||
rawFree(@ptrCast(p));
|
||||
return null;
|
||||
}
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||
const old_payload = block.size - header_size;
|
||||
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||
rawFree(@ptrCast(p));
|
||||
return @ptrCast(np);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||
// In-place iff the new payload still fits the current block.
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||
return new_len + header_size <= block.size;
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,220 @@
|
||||
//! User-space input helpers: the client and publisher sides of the input service, so a
|
||||
//! program listening for input events — or a driver broadcasting them — doesn't hand-roll
|
||||
//! the IPC. Layered over `ipc` (endpoints, capability passing, `send`) and the shared
|
||||
//! `input-protocol` wire format, the way `device.zig` layers over the raw `device_*` calls.
|
||||
//! See system/services/input/input.zig.
|
||||
//!
|
||||
//! The service carries several device classes (keyboard, mouse, joystick/gamepad). A
|
||||
//! **source** publishes its class with the matching method:
|
||||
//! var source = input.connectSource() orelse return;
|
||||
//! _ = source.publishKeyboardEvent(.{ .kind = ..., .keycode = ..., ... });
|
||||
//! _ = source.publishMouseEvent(.{ ... });
|
||||
//! _ = source.publishJoystickEvent(.{ ... });
|
||||
//!
|
||||
//! A **subscriber** either takes one class with a typed helper —
|
||||
//! var keys = input.subscribeKeyboard() orelse return;
|
||||
//! while (true) { const key = keys.next() orelse continue; ... }
|
||||
//! — or takes several at once and inspects the tagged envelope:
|
||||
//! var listener = input.subscribeAll() orelse return;
|
||||
//! while (true) {
|
||||
//! const event = listener.next() orelse continue;
|
||||
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||
//! }
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("input-protocol");
|
||||
|
||||
pub const DeviceKind = protocol.DeviceKind;
|
||||
pub const InputEvent = protocol.InputEvent;
|
||||
pub const KeyEvent = protocol.KeyEvent;
|
||||
pub const MouseEvent = protocol.MouseEvent;
|
||||
pub const JoystickEvent = protocol.JoystickEvent;
|
||||
pub const EventKind = protocol.EventKind;
|
||||
pub const MouseEventKind = protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||
pub const Keycode = protocol.Keycode;
|
||||
|
||||
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||
/// input.device_mouse)`.
|
||||
pub const device_keyboard = protocol.device_keyboard;
|
||||
pub const device_mouse = protocol.device_mouse;
|
||||
pub const device_joystick = protocol.device_joystick;
|
||||
pub const device_all = protocol.device_all;
|
||||
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||
fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- subscribing ------------------------------------------------------------
|
||||
|
||||
/// A subscription to the input service: our own endpoint, which the service pushes events
|
||||
/// to. `next` returns each event as a tagged `InputEvent`; use `asKeyboard`/`asMouse`/
|
||||
/// `asJoystick` to decode. Created with `subscribe`/`subscribeAll`; for a single device
|
||||
/// class prefer the typed helpers (`subscribeKeyboard`, ...), which return decoded events.
|
||||
pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
receive: [protocol.event_size]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||
/// (there should be none), so callers can loop.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
/// Subscribe to every input class (keyboard, mouse, joystick) on one stream.
|
||||
pub fn subscribeAll() ?Subscriber {
|
||||
return subscribe(device_all);
|
||||
}
|
||||
|
||||
/// A subscriber filtered to keyboard events, whose `next` returns a decoded `KeyEvent`.
|
||||
pub const KeyboardSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *KeyboardSubscriber) ?KeyEvent {
|
||||
return (self.inner.next() orelse return null).asKeyboard();
|
||||
}
|
||||
};
|
||||
|
||||
/// A subscriber filtered to mouse events, whose `next` returns a decoded `MouseEvent`.
|
||||
pub const MouseSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *MouseSubscriber) ?MouseEvent {
|
||||
return (self.inner.next() orelse return null).asMouse();
|
||||
}
|
||||
};
|
||||
|
||||
/// A subscriber filtered to joystick/gamepad events, whose `next` returns a decoded
|
||||
/// `JoystickEvent`.
|
||||
pub const JoystickSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *JoystickSubscriber) ?JoystickEvent {
|
||||
return (self.inner.next() orelse return null).asJoystick();
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to keyboard events only; `next` returns decoded `KeyEvent`s.
|
||||
pub fn subscribeKeyboard() ?KeyboardSubscriber {
|
||||
return .{ .inner = subscribe(device_keyboard) orelse return null };
|
||||
}
|
||||
|
||||
/// Subscribe to mouse events only; `next` returns decoded `MouseEvent`s.
|
||||
pub fn subscribeMouse() ?MouseSubscriber {
|
||||
return .{ .inner = subscribe(device_mouse) orelse return null };
|
||||
}
|
||||
|
||||
/// Subscribe to joystick/gamepad events only; `next` returns decoded `JoystickEvent`s.
|
||||
pub fn subscribeJoystick() ?JoystickSubscriber {
|
||||
return .{ .inner = subscribe(device_joystick) orelse return null };
|
||||
}
|
||||
|
||||
// --- publishing -------------------------------------------------------------
|
||||
|
||||
/// A connection to the input service for a source (a keyboard/mouse/joystick driver) that
|
||||
/// publishes events. Each `publish*Event` is a short synchronous call the service answers
|
||||
/// at once; its own fan-out to subscribers is asynchronous, so publishing never blocks on
|
||||
/// a slow subscriber.
|
||||
pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
pub fn publishKeyboardEvent(self: Publisher, event: KeyEvent) bool {
|
||||
return self.publish(InputEvent.fromKeyboard(event));
|
||||
}
|
||||
/// Broadcast a mouse event to every subscriber that took mouse events.
|
||||
pub fn publishMouseEvent(self: Publisher, event: MouseEvent) bool {
|
||||
return self.publish(InputEvent.fromMouse(event));
|
||||
}
|
||||
/// Broadcast a joystick/gamepad event to every subscriber that took joystick events.
|
||||
pub fn publishJoystickEvent(self: Publisher, event: JoystickEvent) bool {
|
||||
return self.publish(InputEvent.fromJoystick(event));
|
||||
}
|
||||
};
|
||||
|
||||
/// Connect to the input service as an event source, waiting for it to come up. Returns a
|
||||
/// `Publisher`, or null if the service never registered.
|
||||
pub fn connectSource() ?Publisher {
|
||||
return .{ .service = lookupService() orelse return null };
|
||||
}
|
||||
|
||||
// --- synthetic scaffolding --------------------------------------------------
|
||||
|
||||
/// Synthetic key events, shared by the demo source and the keyboard driver's placeholder
|
||||
/// stream while real scancode decoding is still a follow-up. `step` rolls through A..E,
|
||||
/// emitting for each key a `key_down`, then a `key_press` carrying the character, then a
|
||||
/// `key_up`. Scaffolding, not wire protocol — hence it lives with the helpers.
|
||||
pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||
const Key = struct { code: Keycode, character: u32 };
|
||||
const keys = [_]Key{
|
||||
.{ .code = .a, .character = 'A' },
|
||||
.{ .code = .b, .character = 'B' },
|
||||
.{ .code = .c, .character = 'C' },
|
||||
.{ .code = .d, .character = 'D' },
|
||||
.{ .code = .e, .character = 'E' },
|
||||
};
|
||||
const key = keys[(step / 3) % keys.len];
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(EventKind.key_down), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||
1 => .{ .kind = @intFromEnum(EventKind.key_press), .keycode = @intFromEnum(key.code), .character = key.character, .modifiers = 0 },
|
||||
else => .{ .kind = @intFromEnum(EventKind.key_up), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/// Synthetic mouse events (placeholder until real PS/2 packet decoding). `step` alternates
|
||||
/// a small diagonal motion with a left-button click.
|
||||
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/// Synthetic joystick/gamepad events (placeholder until a real controller driver). `step`
|
||||
/// sweeps axis 0 and toggles button 0.
|
||||
pub fn syntheticJoystickEvent(step: usize) JoystickEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(JoystickEventKind.axis), .control = 0, .value = 16384, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(JoystickEventKind.button_down), .control = 0, .value = 0, .buttons = 1 },
|
||||
else => .{ .kind = @intFromEnum(JoystickEventKind.button_up), .control = 0, .value = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,194 @@
|
||||
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||
//! added with the first server binary.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
|
||||
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||
/// layer their own wire format on top of the bytes a call carries.
|
||||
pub const Message = extern struct {
|
||||
tag: u64 = 0,
|
||||
a: u64 = 0,
|
||||
b: u64 = 0,
|
||||
c: u64 = 0,
|
||||
};
|
||||
|
||||
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Create a new endpoint owned by this process; returns its handle.
|
||||
pub fn createIpcEndpoint() ?Handle {
|
||||
const r = sc.systemCall0(.create_ipc_endpoint);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
pub const CallError = error{Failed};
|
||||
|
||||
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||
/// an endpoint the server sent back (e.g. a per-device channel), or null.
|
||||
pub const Reply = struct {
|
||||
len: usize,
|
||||
cap: ?Handle,
|
||||
};
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`,
|
||||
/// optionally handing the server a capability (`send_cap`) and receiving one back.
|
||||
/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a
|
||||
/// private per-device endpoint back in `.cap`. Two return values (reply length in rax,
|
||||
/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply
|
||||
/// capacity, arg #4; out: the received handle).
|
||||
pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply {
|
||||
var rax: usize = undefined;
|
||||
var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[r8] "+{r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(message.ptr)),
|
||||
[a2] "{rdx}" (message.len),
|
||||
[a3] "{r10}" (@intFromPtr(reply.ptr)),
|
||||
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return error.Failed;
|
||||
return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||
/// Returns the reply length. The common case: no capability passed either way.
|
||||
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||
return (try callCap(h, message, reply, null)).len;
|
||||
}
|
||||
|
||||
/// Post `message` to endpoint `h`'s asynchronous queue and return immediately — no
|
||||
/// rendezvous, no reply, no blocking. The receiver picks it up through `replyWait` as a
|
||||
/// buffered message (`Received.isMessage`). Unlike `call`, this **cannot hang on a dead
|
||||
/// or slow peer**, which is why a broadcaster (the input service) delivers events this
|
||||
/// way. The payload must fit an endpoint slot (64 bytes); a full queue drops the oldest
|
||||
/// message. Returns false on failure (bad handle, oversized payload, bad buffer).
|
||||
pub fn send(h: Handle, message: []const u8) bool {
|
||||
return !failed(sc.systemCall3(.ipc_send, h, @intFromPtr(message.ptr), message.len));
|
||||
}
|
||||
|
||||
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||
/// GSI. See `isNotification`.
|
||||
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||
/// landing (`system.timerOnce`).
|
||||
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||
pub const notify_exit_bit: u64 = abi.notify_exit_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the wake-up is a **buffered message** — a payload
|
||||
/// posted with `send` (`ipc_send`) — rather than a bare device interrupt or child-exit
|
||||
/// notice. The payload is in the `replyWait` receive buffer (`Received.len` bytes); the
|
||||
/// low bits of the badge carry the sender's task id. See `Received.isMessage`.
|
||||
pub const notify_message_bit: u64 = abi.notify_message_bit;
|
||||
|
||||
/// The result of a `replyWait`: the request length, the sender's badge (a task id, or
|
||||
/// an IRQ notification if the high bit is set), and any capability the request carried.
|
||||
pub const Received = struct {
|
||||
len: usize,
|
||||
badge: u64,
|
||||
cap: ?Handle,
|
||||
|
||||
/// True if this wake-up was an asynchronous notification (a device interrupt
|
||||
/// or a child-exit notice), not a client request. An event loop branches on
|
||||
/// this; there is no reply owed on the notification path.
|
||||
pub fn isNotification(self: Received) bool {
|
||||
return self.badge & notify_badge_bit != 0;
|
||||
}
|
||||
|
||||
/// True if this wake-up tells of a supervised child's end — the notification
|
||||
/// requested by passing an exit endpoint to `system.spawnSupervised`.
|
||||
pub fn isChildExit(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_exit_bit != 0;
|
||||
}
|
||||
|
||||
/// True if this wake-up is a **buffered message** posted with `send` (`ipc_send`):
|
||||
/// there is a payload in the receive buffer (`self.len` bytes) and no reply is owed.
|
||||
/// The subscriber side of a broadcast branches on this.
|
||||
pub fn isMessage(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_message_bit != 0;
|
||||
}
|
||||
|
||||
/// The task id of whoever posted a buffered message, meaningful only when
|
||||
/// Whether this arrival is a signal notification — decode the set with
|
||||
/// `process.signalsFrom(badge)`.
|
||||
pub fn isSignal(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||
}
|
||||
|
||||
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||
pub fn isTimer(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||
}
|
||||
|
||||
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||
pub fn senderTaskId(self: Received) u32 {
|
||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||
}
|
||||
|
||||
/// The interrupt source (a GSI), meaningful only when `isNotification` and
|
||||
/// not `isChildExit`.
|
||||
pub fn source(self: Received) u64 {
|
||||
return self.badge & ~notify_badge_bit;
|
||||
}
|
||||
|
||||
/// The ended child's process id, meaningful only when `isChildExit`.
|
||||
pub fn childProcessId(self: Received) u32 {
|
||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit));
|
||||
}
|
||||
};
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||
/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received
|
||||
/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length,
|
||||
/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle).
|
||||
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge
|
||||
var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
[r8] "+{r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||
//! the argv contract) and the process end of the lifecycle
|
||||
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||
//! no data on freestanding targets, so the type is danos's own.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// Everything a program receives at entry. Passed to
|
||||
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||
/// `pub fn main() void`. An `environment` field is added here once the kernel
|
||||
/// passes a non-empty envp (today it is always empty — see docs/sysv.md).
|
||||
pub const Init = struct {
|
||||
arguments: Arguments,
|
||||
};
|
||||
|
||||
/// The process arguments (argc/argv), parsed from the kernel-built System V
|
||||
/// entry block. The bytes live in the entry block at the top of the stack page,
|
||||
/// NUL-terminated, valid for the process's lifetime.
|
||||
pub const Arguments = struct {
|
||||
/// argc — at least 1: argument 0 is the path or name this binary was
|
||||
/// spawned as.
|
||||
count: usize,
|
||||
/// The argv pointers in the entry block (NULL-terminated after `count`
|
||||
/// entries).
|
||||
vector: [*]const [*:0]const u8,
|
||||
|
||||
/// Argument `index` (0 = the program's own path/name), or null if out of
|
||||
/// range.
|
||||
pub fn get(arguments: Arguments, index: usize) ?[:0]const u8 {
|
||||
if (index >= arguments.count) return null;
|
||||
return std.mem.span(arguments.vector[index]);
|
||||
}
|
||||
|
||||
pub fn iterate(arguments: Arguments) Iterator {
|
||||
return .{ .arguments = arguments };
|
||||
}
|
||||
|
||||
pub const Iterator = struct {
|
||||
arguments: Arguments,
|
||||
index: usize = 0,
|
||||
|
||||
pub fn next(iterator: *Iterator) ?[:0]const u8 {
|
||||
const argument = iterator.arguments.get(iterator.index) orelse return null;
|
||||
iterator.index += 1;
|
||||
return argument;
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||
pub const ExitReason = abi.ExitReason;
|
||||
|
||||
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||
/// kernel records the reason before it posts the notification, so this never
|
||||
/// races it. Returns null for an id that never lived, is still alive, was
|
||||
/// evicted from the kernel's bounded record, or is not this process's child
|
||||
/// (the same authority gate as `kill`).
|
||||
pub fn exitReason(id: u32) ?ExitReason {
|
||||
const r = sc.systemCall1(.process_exit_reason, id);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @enumFromInt(r);
|
||||
}
|
||||
|
||||
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||
/// `system.kill`, unhandleable by definition).
|
||||
pub const Signal = abi.Signal;
|
||||
|
||||
/// The coalesced set of signals one notification delivered: two pending
|
||||
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||
pub const SignalSet = struct {
|
||||
pending: u32,
|
||||
|
||||
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||
pub fn bindSignals(endpoint: usize) bool {
|
||||
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||
}
|
||||
|
||||
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||
/// a signal notification.
|
||||
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||
}
|
||||
|
||||
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||
/// non-blocking, always — a statement, not a conversation.
|
||||
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||
}
|
||||
|
||||
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||
/// concurrent traffic implements the same sequence inside its own event loop
|
||||
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
_ = sendSignal(id, .terminate);
|
||||
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||
var receive: [8]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||
}
|
||||
_ = system.kill(id);
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||
/// services: release what the dead client held — file handles, subscriptions —
|
||||
/// because a service must never depend on clients cleaning up after themselves
|
||||
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||
pub fn subscribeExits(endpoint: usize) bool {
|
||||
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by runtime.start) and the panic handler — and pulls in the
|
||||
//! `_start` entry shim. A program therefore only defines `pub fn main`; nothing
|
||||
//! else is required in its source file.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by runtime.start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (runtime.start.panic).
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary only defines a `pub fn main() void` or
|
||||
//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`).
|
||||
//! The panic handler and the `_start` entry pull live in the shared compilation
|
||||
//! root, library/runtime/root.zig, which build.zig wires around every program —
|
||||
//! nothing to declare per source file.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
/// The VFS wire protocol (shared with the VFS server).
|
||||
pub const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||
pub const power_protocol = @import("power-protocol");
|
||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||
/// service. See library/runtime/input.zig and system/services/input/.
|
||||
pub const input = @import("input.zig");
|
||||
/// The input wire protocol (shared with the input service and its clients).
|
||||
pub const input_protocol = @import("input-protocol");
|
||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||
pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shm.zig
|
||||
/// and docs/display-v2.md.
|
||||
pub const shm = @import("shm.zig");
|
||||
|
||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||
pub const usb = @import("usb.zig");
|
||||
|
||||
/// Block-device client: read/write a block device (a USB stick, via
|
||||
/// usb-storage). See library/runtime/block.zig.
|
||||
pub const block = @import("block.zig");
|
||||
|
||||
/// Display-service client: query the mode, and (from D3) create layers, draw, and
|
||||
/// present frames. See library/runtime/display.zig and system/services/display/.
|
||||
pub const display = @import("display.zig");
|
||||
/// The display wire protocol (shared with the display service and its clients).
|
||||
pub const display_protocol = @import("display-protocol");
|
||||
/// The scanout wire protocol: the compositor's present channel to a native scanout driver
|
||||
/// (virtio-gpu). See system/services/display/scanout-protocol.zig and docs/display-v2.md.
|
||||
pub const scanout_protocol = @import("scanout-protocol");
|
||||
|
||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||
/// layer danos programs use directly, and where the operations that later become
|
||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||
pub const fs = @import("fs.zig");
|
||||
|
||||
/// Re-exported so the root shim (root.zig) can install it as the panic handler.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||
pub const process = @import("process.zig");
|
||||
|
||||
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI
|
||||
/// (docs/threading.md). A binary must be built multi-threaded to spawn.
|
||||
pub const Thread = @import("thread.zig").Thread;
|
||||
|
||||
/// The service harness: one replyWait loop folding requests, signals, and
|
||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||
pub const service = @import("service.zig");
|
||||
|
||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||
pub const allocator = heap.allocator;
|
||||
@@ -0,0 +1,83 @@
|
||||
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||
//! folds protocol requests, signals, and subscribed notifications into
|
||||
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||
//! is satisfied by construction and a service author writes domain logic only.
|
||||
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||
//! messages.
|
||||
//!
|
||||
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||
//! service author to implement — a wedged service simply fails to answer, which
|
||||
//! is the diagnosis (see docs/ipc.md).
|
||||
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Called once with the service's endpoint before the loop starts — the
|
||||
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||
/// Return false to abort startup (the process exits).
|
||||
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||
/// One protocol request from `sender` (a task id): write the reply into
|
||||
/// `reply`, return its length. `capability` is the handle the request
|
||||
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||
/// endpoint). The zero-length ping never reaches this.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
/// The reload signal. Default: ignored.
|
||||
on_reload: ?*const fn () void = null,
|
||||
/// The terminate signal, called before the loop returns. The clean exit is
|
||||
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||
/// arrives with no warning; this is for graceful extras only).
|
||||
on_terminate: ?*const fn () void = null,
|
||||
/// Publish the endpoint under a well-known service id at startup.
|
||||
service: ?abi.ServiceId = null,
|
||||
};
|
||||
|
||||
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||
/// (a service passes its protocol's message maximum).
|
||||
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||
if (callbacks.service) |id| {
|
||||
if (!ipc.register(id, endpoint)) return;
|
||||
}
|
||||
_ = process.bindSignals(endpoint);
|
||||
if (callbacks.init) |initialise| {
|
||||
if (!initialise(endpoint)) return;
|
||||
}
|
||||
|
||||
var reply_buffer: [maximum_message]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
var receive: [maximum_message]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||
if (got.isNotification()) {
|
||||
reply_len = 0; // nothing owed for a notification
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.reload)) {
|
||||
if (callbacks.on_reload) |onReload| onReload();
|
||||
}
|
||||
if (signals.has(.terminate)) {
|
||||
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||
return; // the loop's return IS the clean exit
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
if (got.len == 0) {
|
||||
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||
continue;
|
||||
}
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
//! User-space shared memory: `shm_create` / `shm_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shm_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
}
|
||||
|
||||
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||
/// another process as an `ipc_call` send_cap.
|
||||
pub const Region = struct {
|
||||
ptr: [*]u8,
|
||||
handle: ipc.Handle,
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
|
||||
/// so this is a hand-written stub like `dma.alloc`.
|
||||
pub fn create(len: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: the capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shm_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||
}
|
||||
|
||||
/// Map the shared region named by a capability `handle` this process received (via an
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shm_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
|
||||
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shm_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||
//! `entry = _start` in build.zig); the shared compilation root, root.zig, forces
|
||||
//! this file to be analysed with `comptime { _ = &runtime.start._start; }`, so
|
||||
//! the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||
/// safety net if `rt_start` ever returns.
|
||||
pub export fn _start() callconv(.naked) noreturn {
|
||||
asm volatile (
|
||||
\\mov %%rsp, %%rdi
|
||||
\\call rt_start
|
||||
\\ud2
|
||||
);
|
||||
}
|
||||
|
||||
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||
/// block. Build the `process.Init` from it and dispatch to the program's `main`,
|
||||
/// whose signature is inspected at comptime. The heap is lazy (first alloc grows
|
||||
/// it), so there is no other runtime init to order here.
|
||||
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
const init: process.Init = .{ .arguments = .{
|
||||
.count = stack[0],
|
||||
.vector = @ptrCast(stack + 1),
|
||||
} };
|
||||
system.exit(callMain(init));
|
||||
}
|
||||
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||
fn callMain(init: process.Init) u8 {
|
||||
const root = @import("root"); // root.zig, re-exporting the program's main
|
||||
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||
|
||||
const call_arguments = switch (main_information.params.len) {
|
||||
0 => .{},
|
||||
1 => arguments: {
|
||||
const Parameter = main_information.params[0].type orelse
|
||||
@compileError("main's parameter must be runtime.process.Init (not anytype)");
|
||||
if (Parameter != process.Init)
|
||||
@compileError("main's parameter must be runtime.process.Init, found " ++ @typeName(Parameter));
|
||||
break :arguments .{init};
|
||||
},
|
||||
else => @compileError("main takes no parameters or a single runtime.process.Init"),
|
||||
};
|
||||
|
||||
const ReturnType = main_information.return_type.?;
|
||||
switch (@typeInfo(ReturnType)) {
|
||||
.noreturn => @call(.auto, root.main, call_arguments),
|
||||
.void => {
|
||||
@call(.auto, root.main, call_arguments);
|
||||
return 0;
|
||||
},
|
||||
.int => {
|
||||
if (ReturnType != u8)
|
||||
@compileError("main's integer return type must be u8, found " ++ @typeName(ReturnType));
|
||||
return @call(.auto, root.main, call_arguments);
|
||||
},
|
||||
.error_union => {
|
||||
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||
var buffer: [128]u8 = undefined;
|
||||
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||
_ = system.write(line);
|
||||
return 1; // distinct from panic's 127
|
||||
};
|
||||
if (@TypeOf(payload) == void) return 0;
|
||||
if (@TypeOf(payload) == u8) return payload;
|
||||
@compileError("main's error-union payload must be void or u8, found " ++ @typeName(@TypeOf(payload)));
|
||||
},
|
||||
else => @compileError("main must return void, noreturn, u8, !void, or !u8, found " ++ @typeName(ReturnType)),
|
||||
}
|
||||
}
|
||||
|
||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
system.exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,67 @@
|
||||
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||
//!
|
||||
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||
//! instruction, so the kernel reads the 4th argument from r10.
|
||||
|
||||
const abi = @import("abi");
|
||||
const SystemCall = abi.SystemCall;
|
||||
|
||||
pub inline fn systemCall0(n: SystemCall) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
[a4] "{r8}" (a4),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||
//! source of truth shared with the kernel dispatcher.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||
/// declare its snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
pub fn sleep(ms: usize) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||
/// and restart backoff are built from.
|
||||
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||
///
|
||||
/// const deadline = clock() + timeout_ns;
|
||||
/// while (clock() < deadline) { ... }
|
||||
pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
|
||||
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
|
||||
/// filesystem stamps as a file's modification time. Formatting it into a calendar
|
||||
/// date/timezone is user-space policy layered on top.
|
||||
pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Copy bytes out of the kernel's in-memory diagnostic log — the accumulated
|
||||
/// stream of everything `write` (and the kernel itself) has emitted — starting at
|
||||
/// `offset`, into `out`. Returns the number of bytes copied (0 at end of buffer).
|
||||
/// A program reads the whole log by looping from offset 0, advancing by the return
|
||||
/// value, until it gets 0. This is how the boot log is persisted to disk on a
|
||||
/// headless/real machine where serial output is otherwise lost.
|
||||
pub fn klogRead(offset: usize, out: []u8) usize {
|
||||
return sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||
/// process, returning the child's process id (or null on failure). The child's
|
||||
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||
/// POSIX layer later).
|
||||
pub fn spawn(name: []const u8) ?u32 {
|
||||
return spawnSupervised(name, &.{}, null);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||
return spawnSupervised(name, arguments, null);
|
||||
}
|
||||
|
||||
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||
/// null on failure.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
if (i != 0) {
|
||||
if (len >= blob.len) return null;
|
||||
blob[len] = 0;
|
||||
len += 1;
|
||||
}
|
||||
if (len + argument.len > blob.len) return null;
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||
/// The primitive `ps` is built on.
|
||||
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||
pub fn isProcessRunning(name: []const u8) bool {
|
||||
var table: [32]ProcessDescriptor = undefined;
|
||||
const total = processes(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||
/// another core dies at its next system call or timer tick. True means the kill
|
||||
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||
/// at spawn) confirms completion.
|
||||
pub fn kill(id: u32) bool {
|
||||
return sc.systemCall1(.process_kill, id) == 0;
|
||||
}
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||
/// memory and return the base virtual address. On failure returns a value in the
|
||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
@@ -0,0 +1,484 @@
|
||||
//! `runtime.Thread` — threads for danos, shaped like Zig's `std.Thread` but built on
|
||||
//! danos's private thread ABI (docs/threading.md). Several tasks share one address
|
||||
//! space; `spawn` starts one, the kernel delivers the closure pointer in the new
|
||||
//! thread's rdi, a plain Zig trampoline runs the user function and calls `thread_exit`,
|
||||
//! and `join` blocks on the thread's exit notification. See docs/threading.md for why
|
||||
//! this mirrors `std.Thread`'s API rather than being the literal type.
|
||||
//!
|
||||
//! The closure (the function's captured args) lives at the **top of the thread's own
|
||||
//! stack**, not the heap — each thread's stack is private, so there is no shared-heap
|
||||
//! concurrency in the spawn/join machinery (the runtime heap is not yet thread-safe).
|
||||
//! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||
/// machines can be exercised on the host against `std.Thread.Futex` (docs/threading-plan.md
|
||||
/// M11), while the danos build uses the futex syscalls.
|
||||
const on_danos = builtin.os.tag == .freestanding;
|
||||
|
||||
/// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages.
|
||||
pub const default_stack_size: usize = 64 * 1024;
|
||||
|
||||
/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the
|
||||
/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10.
|
||||
const tls_block_size: usize = 64;
|
||||
|
||||
pub const Thread = struct {
|
||||
/// The kernel task id of the spawned thread — what `join` waits on.
|
||||
tid: u32,
|
||||
/// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`).
|
||||
stack_base: usize,
|
||||
stack_size: usize,
|
||||
|
||||
pub const Id = u32;
|
||||
|
||||
pub const SpawnConfig = struct {
|
||||
/// Bytes of stack, rounded up to whole pages by the kernel's mmap.
|
||||
stack_size: usize = default_stack_size,
|
||||
};
|
||||
|
||||
pub const SpawnError = error{
|
||||
/// The kernel refused the thread, the stack mmap failed, or no endpoint was free.
|
||||
SystemResources,
|
||||
};
|
||||
|
||||
/// Start `function(args...)` on a new thread sharing this address space. Mirrors
|
||||
/// `std.Thread.spawn`. The thread's return value is discarded (as in `std.Thread`);
|
||||
/// return data through shared state.
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread {
|
||||
const Args = @TypeOf(args);
|
||||
const Closure = struct {
|
||||
tls_base: usize,
|
||||
args: Args,
|
||||
/// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this
|
||||
/// thread's TLS pointer, runs the user function, then ends the thread.
|
||||
fn entry(self_addr: usize) callconv(.c) noreturn {
|
||||
const self: *@This() = @ptrFromInt(self_addr);
|
||||
setThreadPointer(self.tls_base); // per-thread thread pointer before any user code
|
||||
@call(.auto, function, self.args);
|
||||
exitThread();
|
||||
}
|
||||
};
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||
// scratch for user TLS), then the stack proper (rsp starts below the TLS block, so
|
||||
// the growing stack never overwrites either).
|
||||
var closure_addr = (base + config.stack_size) - @sizeOf(Closure);
|
||||
closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down
|
||||
|
||||
const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15);
|
||||
const tls: [*]usize = @ptrFromInt(tls_base);
|
||||
tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects
|
||||
|
||||
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||
closure.* = .{ .tls_base = tls_base, .args = args };
|
||||
|
||||
var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block
|
||||
stack_top -= 8; // ...then rsp % 16 == 8 at the C entry
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||
}
|
||||
|
||||
/// Block until this thread finishes, then reclaim its stack. Mirrors
|
||||
/// `std.Thread.join`. The exit endpoint is private to this thread, so the first
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
/// reclaimed at process exit (docs/threading-plan.md M3 — kernel-reaper stack reclaim
|
||||
/// for detached threads is a later refinement). Mirrors `std.Thread.detach`.
|
||||
pub fn detach(self: Thread) void {
|
||||
_ = self;
|
||||
}
|
||||
|
||||
/// The calling thread's id (its kernel task id). Mirrors `std.Thread.getCurrentId`.
|
||||
pub fn getCurrentId() Id {
|
||||
return @intCast(sc.systemCall0(.thread_self));
|
||||
}
|
||||
|
||||
/// The dense 0-based index of the core the calling thread is running on. A danos
|
||||
/// extension beyond `std.Thread`, used to observe genuine cross-core parallelism.
|
||||
pub fn currentCore() Id {
|
||||
return @intCast(sc.systemCall0(.current_core));
|
||||
}
|
||||
|
||||
/// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the
|
||||
/// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the
|
||||
/// kernel (no busy-wait), so an idle core still halts (docs/halting.md).
|
||||
pub const Futex = struct {
|
||||
/// Block while `ptr.* == expect`. Returns when woken by `wake`, or promptly if
|
||||
/// the value already differs (safe against spurious returns, as in std): the
|
||||
/// caller re-checks its condition in a loop.
|
||||
pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void {
|
||||
if (comptime on_danos) {
|
||||
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||
} else {
|
||||
// Host unit-test mock: spin+yield until the value changes (`wake` is a
|
||||
// no-op — the callers re-check their condition in a loop anyway). Correct,
|
||||
// if busy; fine for the state-machine tests.
|
||||
while (ptr.load(.acquire) == expect) std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first.
|
||||
pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void {
|
||||
if (comptime on_danos) {
|
||||
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||
} else {
|
||||
var spins: u64 = 0;
|
||||
const limit = timeout_ns / 1000 + 1;
|
||||
while (ptr.load(.acquire) == expect) : (spins += 1) {
|
||||
if (spins >= limit) return error.Timeout;
|
||||
std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake up to `max_waiters` threads blocked on `ptr`.
|
||||
pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void {
|
||||
if (comptime on_danos) {
|
||||
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||
} else {
|
||||
// host mock: spin-waiters re-check their condition, so no wake is needed.
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// A mutual-exclusion lock, `std.Thread.Mutex`-shaped. The classic three-state
|
||||
/// futex mutex (unlocked / locked / contended): the fast path is a single CAS, and
|
||||
/// only a contended lock ever enters the kernel.
|
||||
pub const Mutex = struct {
|
||||
state: std.atomic.Value(u32) = std.atomic.Value(u32).init(unlocked),
|
||||
|
||||
const unlocked: u32 = 0;
|
||||
const locked: u32 = 1;
|
||||
const contended: u32 = 2;
|
||||
|
||||
/// Try to take the lock without blocking; returns whether it was acquired.
|
||||
pub fn tryLock(m: *Mutex) bool {
|
||||
return m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) == null;
|
||||
}
|
||||
|
||||
/// Acquire the lock, blocking in the kernel while it is contended.
|
||||
pub fn lock(m: *Mutex) void {
|
||||
if (m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) != null) m.lockSlow();
|
||||
}
|
||||
|
||||
fn lockSlow(m: *Mutex) void {
|
||||
@branchHint(.cold);
|
||||
// Mark the lock contended and take it as soon as it falls unlocked; park on
|
||||
// the futex while it stays contended. Marking contended may cause a spurious
|
||||
// wake on unlock (harmless), never a missed one.
|
||||
while (m.state.swap(contended, .acquire) != unlocked) {
|
||||
Futex.wait(&m.state, contended);
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the lock; wake one waiter if the lock was contended.
|
||||
pub fn unlock(m: *Mutex) void {
|
||||
if (m.state.swap(unlocked, .release) == contended) Futex.wake(&m.state, 1);
|
||||
}
|
||||
};
|
||||
|
||||
/// A condition variable, `std.Thread.Condition`-shaped. Spurious wakeups are
|
||||
/// allowed — always wait in a predicate loop with the mutex held. Built on a futex
|
||||
/// sequence counter: a waiter samples the seq, drops the mutex, and parks until the
|
||||
/// seq changes (a signal that races the unlock bumps the seq, so it is not missed).
|
||||
pub const Condition = struct {
|
||||
seq: std.atomic.Value(u32) = std.atomic.Value(u32).init(0),
|
||||
|
||||
/// Atomically release `mutex` and block until signalled, then re-acquire it.
|
||||
pub fn wait(c: *Condition, mutex: *Mutex) void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
Futex.wait(&c.seq, seq);
|
||||
mutex.lock();
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first. The
|
||||
/// mutex is re-acquired either way.
|
||||
pub fn timedWait(c: *Condition, mutex: *Mutex, timeout_ns: u64) error{Timeout}!void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
const timed_out = if (Futex.timedWait(&c.seq, seq, timeout_ns)) |_| false else |_| true;
|
||||
mutex.lock();
|
||||
if (timed_out) return error.Timeout;
|
||||
}
|
||||
|
||||
/// Wake one waiter.
|
||||
pub fn signal(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, 1);
|
||||
}
|
||||
|
||||
/// Wake all waiters.
|
||||
pub fn broadcast(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, std.math.maxInt(u32));
|
||||
}
|
||||
};
|
||||
|
||||
/// A counting semaphore, `std.Thread.Semaphore`-shaped: a permit count guarded by a
|
||||
/// `Mutex` + `Condition`.
|
||||
pub const Semaphore = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
permits: usize = 0,
|
||||
|
||||
/// Take a permit, blocking until one is available.
|
||||
pub fn wait(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
while (s.permits == 0) s.cond.wait(&s.mutex);
|
||||
s.permits -= 1;
|
||||
}
|
||||
|
||||
/// Return a permit and wake a waiter.
|
||||
pub fn post(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
s.permits += 1;
|
||||
s.cond.signal();
|
||||
}
|
||||
};
|
||||
|
||||
/// A reader/writer lock, `std.Thread.RwLock`-shaped: many concurrent readers OR one
|
||||
/// exclusive writer. Reader-preferring (a steady stream of readers can delay a writer),
|
||||
/// built on `Mutex` + `Condition` over a signed state: `>0` = that many readers hold
|
||||
/// it, `-1` = a writer holds it, `0` = free.
|
||||
pub const RwLock = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
state: i64 = 0,
|
||||
|
||||
/// Acquire shared (read) access, blocking while a writer holds the lock.
|
||||
pub fn lockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state < 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state += 1;
|
||||
}
|
||||
|
||||
/// Try to acquire shared access without blocking.
|
||||
pub fn tryLockShared(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state < 0) return false;
|
||||
rw.state += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release shared access; wake a waiting writer once the last reader leaves.
|
||||
pub fn unlockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state -= 1;
|
||||
if (rw.state == 0) rw.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Acquire exclusive (write) access, blocking until no readers or writer remain.
|
||||
pub fn lock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state != 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state = -1;
|
||||
}
|
||||
|
||||
/// Try to acquire exclusive access without blocking.
|
||||
pub fn tryLock(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state != 0) return false;
|
||||
rw.state = -1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release exclusive access; wake all waiters (they re-check their condition).
|
||||
pub fn unlock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state = 0;
|
||||
rw.cond.broadcast();
|
||||
}
|
||||
};
|
||||
|
||||
/// A `std.Thread.WaitGroup`-shaped counter: `start` before spawning work, `finish` as
|
||||
/// each unit completes, `wait` blocks until the count returns to zero.
|
||||
pub const WaitGroup = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
counter: usize = 0,
|
||||
|
||||
/// Register one pending unit of work.
|
||||
pub fn start(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter += 1;
|
||||
}
|
||||
|
||||
/// Mark one unit done; wake waiters if that was the last.
|
||||
pub fn finish(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter -= 1;
|
||||
if (wg.counter == 0) wg.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Block until every started unit has finished.
|
||||
pub fn wait(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
while (wg.counter != 0) wg.cond.wait(&wg.mutex);
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error.
|
||||
fn threadSpawn(entry: usize, stack_top: usize, arg: usize) usize {
|
||||
const exit_endpoint: usize = @intCast(abi.no_cap); // join uses thread_join, not an endpoint
|
||||
return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint);
|
||||
}
|
||||
|
||||
/// The kernel returns a real (small) task id on success and a wrapped `-1` on failure;
|
||||
/// no valid task id ever exceeds a u32.
|
||||
inline fn threadSpawnFailed(ret: usize) bool {
|
||||
return ret > std.math.maxInt(u32);
|
||||
}
|
||||
|
||||
/// End the calling thread. Never returns.
|
||||
fn exitThread() noreturn {
|
||||
_ = sc.systemCall0(.thread_exit);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Set the calling thread's FS base (its user TLS thread pointer).
|
||||
fn setThreadPointer(addr: usize) void {
|
||||
_ = sc.systemCall1(.set_thread_pointer, addr);
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*).
|
||||
fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||
return sc.systemCall3(.futex_wait, addr, expect, timeout_ns);
|
||||
}
|
||||
|
||||
/// futex_wake(addr, count) -> number woken.
|
||||
fn futexWake(addr: usize, count: u32) usize {
|
||||
return sc.systemCall2(.futex_wake, addr, count);
|
||||
}
|
||||
|
||||
// --- host unit tests (docs/threading-plan.md M11) ---------------------------
|
||||
//
|
||||
// These run under `zig build test` on the host: the `Futex` seam above uses
|
||||
// `std.Thread.Futex` off-danos, so the lock/condvar state machines can be exercised by
|
||||
// real host threads. They are never compiled into a danos binary (test blocks only build
|
||||
// under test), so their `std.Thread` use is fine even though `std.Thread` is unavailable
|
||||
// on the freestanding target.
|
||||
|
||||
test "Mutex serialises concurrent increments across host threads" {
|
||||
var m: Thread.Mutex = .{};
|
||||
var counter: u64 = 0;
|
||||
const workers = 8;
|
||||
const per = 20_000;
|
||||
const Ctx = struct {
|
||||
m: *Thread.Mutex,
|
||||
c: *u64,
|
||||
fn run(ctx: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < per) : (i += 1) {
|
||||
ctx.m.lock();
|
||||
ctx.c.* += 1;
|
||||
ctx.m.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
var handles: [workers]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .m = &m, .c = &counter }});
|
||||
for (handles) |h| h.join();
|
||||
try std.testing.expectEqual(@as(u64, workers * per), counter);
|
||||
}
|
||||
|
||||
test "RwLock never lets a reader observe a half-written pair" {
|
||||
var rw: Thread.RwLock = .{};
|
||||
var a: u64 = 0;
|
||||
var b: u64 = 0; // invariant while a lock is held: a == b
|
||||
var stop = std.atomic.Value(bool).init(false);
|
||||
var ok = std.atomic.Value(bool).init(true);
|
||||
|
||||
const Writer = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
stop: *std.atomic.Value(bool),
|
||||
fn run(w: @This()) void {
|
||||
var v: u64 = 1;
|
||||
while (!w.stop.load(.acquire)) : (v +%= 1) {
|
||||
w.rw.lock();
|
||||
w.a.* = v; // update both halves under the exclusive lock...
|
||||
w.b.* = v;
|
||||
w.rw.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
const Reader = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
ok: *std.atomic.Value(bool),
|
||||
fn run(r: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < 200_000) : (i += 1) {
|
||||
r.rw.lockShared();
|
||||
if (r.a.* != r.b.*) r.ok.store(false, .release); // ...so a reader must never see them differ
|
||||
r.rw.unlockShared();
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
var writers: [2]std.Thread = undefined;
|
||||
for (&writers) |*w| w.* = try std.Thread.spawn(.{}, Writer.run, .{Writer{ .rw = &rw, .a = &a, .b = &b, .stop = &stop }});
|
||||
var readers: [4]std.Thread = undefined;
|
||||
for (&readers) |*rd| rd.* = try std.Thread.spawn(.{}, Reader.run, .{Reader{ .rw = &rw, .a = &a, .b = &b, .ok = &ok }});
|
||||
for (readers) |rd| rd.join();
|
||||
stop.store(true, .release);
|
||||
for (writers) |w| w.join();
|
||||
try std.testing.expect(ok.load(.acquire));
|
||||
}
|
||||
|
||||
test "WaitGroup blocks until every started unit finishes" {
|
||||
var wg: Thread.WaitGroup = .{};
|
||||
var done = std.atomic.Value(u32).init(0);
|
||||
const n = 6;
|
||||
const Ctx = struct {
|
||||
wg: *Thread.WaitGroup,
|
||||
done: *std.atomic.Value(u32),
|
||||
fn run(c: @This()) void {
|
||||
_ = c.done.fetchAdd(1, .monotonic);
|
||||
c.wg.finish();
|
||||
}
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) wg.start();
|
||||
var handles: [n]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .wg = &wg, .done = &done }});
|
||||
wg.wait(); // must not return until all n finished
|
||||
try std.testing.expectEqual(@as(u32, n), done.load(.acquire));
|
||||
for (handles) |h| h.join();
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||
//!
|
||||
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||
//!
|
||||
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||
//! CLOCK_REALTIME) layered on top later.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
const nanos_per_micro: u64 = 1_000;
|
||||
const nanos_per_milli: u64 = 1_000_000;
|
||||
const nanos_per_second: u64 = 1_000_000_000;
|
||||
|
||||
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||
/// kernel's millisecond granularity and rounding down could return early.
|
||||
pub const Duration = struct {
|
||||
ns: u64,
|
||||
|
||||
pub fn fromNanos(n: u64) Duration {
|
||||
return .{ .ns = n };
|
||||
}
|
||||
pub fn fromMicros(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_micro };
|
||||
}
|
||||
pub fn fromMillis(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_milli };
|
||||
}
|
||||
pub fn fromSeconds(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_second };
|
||||
}
|
||||
|
||||
pub fn asNanos(d: Duration) u64 {
|
||||
return d.ns;
|
||||
}
|
||||
pub fn asMicros(d: Duration) u64 {
|
||||
return d.ns / nanos_per_micro;
|
||||
}
|
||||
pub fn asMillis(d: Duration) u64 {
|
||||
return d.ns / nanos_per_milli;
|
||||
}
|
||||
pub fn asSeconds(d: Duration) u64 {
|
||||
return d.ns / nanos_per_second;
|
||||
}
|
||||
|
||||
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||
pub fn ceilMillis(d: Duration) u64 {
|
||||
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||
}
|
||||
|
||||
pub fn plus(a: Duration, b: Duration) Duration {
|
||||
return .{ .ns = a.ns +| b.ns };
|
||||
}
|
||||
};
|
||||
|
||||
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||
/// saturate at zero rather than wrap.
|
||||
pub const Instant = struct {
|
||||
ns: u64,
|
||||
|
||||
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||
return .{ .ns = self.ns -| earlier.ns };
|
||||
}
|
||||
|
||||
/// How long since this instant, sampled now.
|
||||
pub fn elapsed(self: Instant) Duration {
|
||||
return now().since(self);
|
||||
}
|
||||
|
||||
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||
pub fn plus(self: Instant, d: Duration) Instant {
|
||||
return .{ .ns = self.ns +| d.ns };
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||
pub fn reached(deadline: Instant) bool {
|
||||
return now().ns >= deadline.ns;
|
||||
}
|
||||
};
|
||||
|
||||
/// The current monotonic time.
|
||||
pub fn now() Instant {
|
||||
return .{ .ns = system.clock() };
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||
/// want a plain integer instead of an `Instant`.
|
||||
pub fn monotonicNanos() u64 {
|
||||
return system.clock();
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||
/// "unavailable" instead of assuming the clock advances.
|
||||
pub fn available() bool {
|
||||
return system.clock() != 0;
|
||||
}
|
||||
|
||||
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||
/// `spin`.
|
||||
pub fn sleep(d: Duration) void {
|
||||
system.sleep(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
system.sleep(ms);
|
||||
}
|
||||
|
||||
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||
/// Prefer `sleep` for anything at or above a millisecond.
|
||||
pub fn spin(d: Duration) void {
|
||||
const deadline = now().plus(d);
|
||||
while (!deadline.reached()) {}
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||
pub fn after(endpoint: usize, d: Duration) bool {
|
||||
return system.timerOnce(endpoint, d.ceilMillis());
|
||||
}
|
||||
|
||||
test "Duration unit conversions round toward zero" {
|
||||
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||
}
|
||||
|
||||
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||
}
|
||||
|
||||
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||
const t0 = Instant{ .ns = 1_000 };
|
||||
const t1 = Instant{ .ns = 4_000 };
|
||||
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||
}
|
||||
|
||||
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
//! USB class-driver client: the helper a keyboard, mouse, or mass-storage driver
|
||||
//! uses to reach its device through the xHCI bus driver, so it never hand-rolls
|
||||
//! the transfer-protocol IPC. Layered over `ipc` and the shared
|
||||
//! `usb-transfer-protocol` wire format, the way `input.zig` layers over the input
|
||||
//! service and `device.zig` over the raw device calls.
|
||||
//!
|
||||
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||
//! if (!usb.helloManager(id)) return; // meet the spawn deadline
|
||||
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||
//! while (true) { ... ipc.replyWait(device.endpoint, ...) ... } // its own loop
|
||||
//!
|
||||
//! Reports are delivered to `device.endpoint` as asynchronous `InterruptReport`
|
||||
//! messages (the class driver runs a bare `replyWait` loop to read them, because
|
||||
//! the service harness drops buffered-message payloads — see service.zig).
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("usb-transfer-protocol");
|
||||
const device_manager = @import("device-manager-protocol");
|
||||
|
||||
pub const Endpoint = protocol.Endpoint;
|
||||
pub const InterruptReport = protocol.InterruptReport;
|
||||
pub const max_report_data = protocol.max_report_data;
|
||||
|
||||
// Endpoint transfer types (EndpointDescriptor attributes), for `findEndpoint`.
|
||||
pub const transfer_type_bulk: u8 = 2;
|
||||
pub const transfer_type_interrupt: u8 = 3;
|
||||
|
||||
/// An opened USB device: the bus endpoint to send requests to, this driver's own
|
||||
/// endpoint that reports arrive on, the device token, and the interface's
|
||||
/// endpoints (so a driver need not re-read the configuration descriptor).
|
||||
pub const Device = struct {
|
||||
bus: ipc.Handle,
|
||||
endpoint: ipc.Handle,
|
||||
token: u64,
|
||||
class: u8,
|
||||
subclass: u8,
|
||||
protocol_code: u8,
|
||||
interface_number: u8,
|
||||
endpoint_count: usize = 0,
|
||||
endpoints: [protocol.max_reported_endpoints]Endpoint = undefined,
|
||||
|
||||
/// The interface's first endpoint of the given transfer type and direction
|
||||
/// (`transfer_type_bulk` / `transfer_type_interrupt`), or null.
|
||||
pub fn findEndpoint(self: *const Device, transfer_type: u8, direction_in: bool) ?Endpoint {
|
||||
for (self.endpoints[0..self.endpoint_count]) |endpoint| {
|
||||
if (endpoint.transfer_type == transfer_type and (endpoint.address & 0x80 != 0) == direction_in) return endpoint;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||
var request = protocol.ControlRequest{
|
||||
.device_token = self.token,
|
||||
.setup = setup,
|
||||
.direction_in = @intFromBool(direction_in),
|
||||
.data_length = @intCast(data.len),
|
||||
};
|
||||
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||
var reply: [@sizeOf(protocol.ControlReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (length < @sizeOf(protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(protocol.ControlReply, reply[0..@sizeOf(protocol.ControlReply)]);
|
||||
if (control_reply.status != 0) return null;
|
||||
const actual = @min(control_reply.actual_length, data.len);
|
||||
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||
return actual;
|
||||
}
|
||||
|
||||
/// A control transfer with no data stage (SET_PROTOCOL, SET_IDLE, ...). The
|
||||
/// `setup` is a bit-cast `usb_abi.Request`.
|
||||
pub fn controlOut(self: *Device, setup: [8]u8) bool {
|
||||
return self.controlTransfer(setup, false, &.{}) != null;
|
||||
}
|
||||
|
||||
/// A device-to-host control transfer, returning the bytes read into `out`.
|
||||
pub fn controlIn(self: *Device, setup: [8]u8, out: []u8) ?usize {
|
||||
return self.controlTransfer(setup, true, out);
|
||||
}
|
||||
|
||||
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||
var request = protocol.InterruptSubscribeRequest{
|
||||
.device_token = self.token,
|
||||
.endpoint_address = endpoint_address,
|
||||
.max_length = max_length,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (length < @sizeOf(protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(protocol.InterruptSubscribeReply, reply[0..@sizeOf(protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
var request = protocol.BulkRequest{
|
||||
.device_token = self.token,
|
||||
.physical_address = physical,
|
||||
.length = length,
|
||||
.endpoint_address = endpoint_address,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.BulkReply)]u8 = undefined;
|
||||
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (replied < @sizeOf(protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(protocol.BulkReply, reply[0..@sizeOf(protocol.BulkReply)]);
|
||||
if (bulk_reply.status != 0) return null;
|
||||
return bulk_reply.actual_length;
|
||||
}
|
||||
};
|
||||
|
||||
/// Look up the USB bus and open the device with the assigned id, handing over a
|
||||
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
|
||||
/// bus is still coming up (a class driver races the bus driver at boot).
|
||||
pub fn open(device_id: u64) ?Device {
|
||||
var attempts: usize = 0;
|
||||
const bus = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
} else return null;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
var request = protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(protocol.OpenReply)]u8 = undefined;
|
||||
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < @sizeOf(protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(protocol.OpenReply, reply[0..@sizeOf(protocol.OpenReply)]);
|
||||
if (open_reply.status != 0) return null;
|
||||
|
||||
var device = Device{
|
||||
.bus = bus,
|
||||
.endpoint = endpoint,
|
||||
.token = open_reply.device_token,
|
||||
.class = open_reply.interface_class,
|
||||
.subclass = open_reply.interface_subclass,
|
||||
.protocol_code = open_reply.interface_protocol,
|
||||
.interface_number = open_reply.interface_number,
|
||||
.endpoint_count = @min(open_reply.endpoint_count, protocol.max_reported_endpoints),
|
||||
};
|
||||
for (0..device.endpoint_count) |index| device.endpoints[index] = open_reply.endpoints[index];
|
||||
return device;
|
||||
}
|
||||
|
||||
/// Hello the device manager as a class driver (Role.device) so a supervised
|
||||
/// spawn meets its hello deadline. Retries while the manager comes up.
|
||||
pub fn helloManager(device_id: u64) bool {
|
||||
var attempts: usize = 0;
|
||||
const manager = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
} else return false;
|
||||
|
||||
const hello = device_manager.Hello{ .role = @intFromEnum(device_manager.Role.device), .device_id = device_id };
|
||||
var reply: [device_manager.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch return false;
|
||||
if (length < device_manager.reply_size) return false;
|
||||
return std.mem.bytesToValue(device_manager.HelloReply, reply[0..device_manager.reply_size]).status == 0;
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
/* Shared link layout for every user binary (init, servers, drivers).
|
||||
*
|
||||
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||
* base, so the entry point comes from e_entry, not the base address.
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||
* permission segment alongside the normal names. */
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
*(.ltext .ltext.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
*(.lrodata .lrodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
*(.ldata .ldata.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space; the loader zeroes the
|
||||
* filesz..memsz gap. */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(.lbss .lbss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||
|
||||
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/input.md)
|
||||
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||
`character` — a keymap — without danos shipping an X11 runtime.
|
||||
|
||||
The layout data comes from the X11 [xkeyboard-config](https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config)
|
||||
database, but it is **compiled to native Zig at build time** rather than parsed at runtime.
|
||||
`tools/make-xkeyboard-config.py` reads the vendored xkb data and emits pure-data tables into
|
||||
`generated/layouts.zig`; `xkeyboard-config.zig` is the hand-written API over them. This is
|
||||
the same build-time-codegen pattern as `tools/make-initial-ramdisk.py`.
|
||||
|
||||
## Using it
|
||||
|
||||
```zig
|
||||
const xkb = @import("xkeyboard-config");
|
||||
|
||||
const m = xkb.map(xkb.us, key_event.keycode, .{ .shift = shift_held, .caps_lock = caps });
|
||||
if (m.character) |ch| { /* a printable Unicode scalar */ }
|
||||
// m.keysym is always set (e.g. an X11 keysym for Return / F1 / a dead key).
|
||||
|
||||
const layout = xkb.byName("gb") orelse xkb.us; // choose a layout by name
|
||||
for (xkb.all) |l| { /* enumerate available layouts */ }
|
||||
```
|
||||
|
||||
`Modifiers` carries `shift`, `caps_lock`, `level3` (AltGr), and `control`. `map` selects the
|
||||
level from the key's XKB *type* (the generated data) and those modifiers (the policy, in
|
||||
`xkeyboard-config.zig`), so data and semantics stay separable.
|
||||
|
||||
Layouts: **us, gb, de, fr, es, dvorak**.
|
||||
|
||||
## Regenerating
|
||||
|
||||
```sh
|
||||
python3 tools/make-xkeyboard-config.py fetch # network: download + vendor the data subset
|
||||
python3 tools/make-xkeyboard-config.py generate # offline: emit generated/layouts.zig
|
||||
# or, from the build:
|
||||
zig build gen-xkeyboard-config
|
||||
```
|
||||
|
||||
- **`fetch`** downloads the pinned xkeyboard-config release (version + sha256 in the script),
|
||||
resolves the `include` graph for the configured layouts, and vendors *only* the symbols
|
||||
files actually reached (plus `keysymdef.h`, `COPYING`, and `PROVENANCE.md`) into `vendor/`.
|
||||
Run it when bumping the version or adding a layout.
|
||||
- **`generate`** is deterministic and offline — same vendored input produces byte-identical
|
||||
output. To add a layout, extend `TARGETS` (and `HID_TO_NAME` if a new physical key is
|
||||
involved), then re-run `fetch` (to vendor any new includes) and `generate`.
|
||||
|
||||
## Scope
|
||||
|
||||
A pragmatic subset, enough for real Latin-script typing:
|
||||
|
||||
- **Group 1 only** — no multi-layout group switching.
|
||||
- **No dead-key / compose composition** — a dead key returns its keysym with no `character`
|
||||
(composing `´` + `e` → `é` is a higher layer's job).
|
||||
- **Curated key types** — the common XKB types (one/two-level, alphabetic, four-level, …);
|
||||
unmapped keys and unknown types fall back to level-by-shift.
|
||||
- **6 layouts** — extend via `TARGETS` as above.
|
||||
|
||||
## Licensing
|
||||
|
||||
xkeyboard-config and `keysymdef.h` (xorgproto) are MIT/X11 licensed. The vendored data
|
||||
subset carries the upstream `vendor/COPYING`, and `vendor/PROVENANCE.md` records the exact
|
||||
version, source URL, and sha256. The generated tables are a derived work under the same terms.
|
||||
File diff suppressed because it is too large
Load Diff
+190
@@ -0,0 +1,190 @@
|
||||
Copyright 1996 by Joseph Moss
|
||||
Copyright (C) 2002-2007 Free Software Foundation, Inc.
|
||||
Copyright (C) Dmitry Golubev <lastguru@mail.ru>, 2003-2004
|
||||
Copyright (C) 2004, Gregory Mokhin <mokhin@bog.msu.ru>
|
||||
Copyright (C) 2006 Erdal Ronahî
|
||||
|
||||
Permission to use, copy, modify, distribute, and sell this software and its
|
||||
documentation for any purpose is hereby granted without fee, provided that
|
||||
the above copyright notice appear in all copies and that both that
|
||||
copyright notice and this permission notice appear in supporting
|
||||
documentation, and that the name of the copyright holder(s) not be used in
|
||||
advertising or publicity pertaining to distribution of the software without
|
||||
specific, written prior permission. The copyright holder(s) makes no
|
||||
representations about the suitability of this software for any purpose. It
|
||||
is provided "as is" without express or implied warranty.
|
||||
|
||||
THE COPYRIGHT HOLDER(S) DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE,
|
||||
INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS, IN NO
|
||||
EVENT SHALL THE COPYRIGHT HOLDER(S) BE LIABLE FOR ANY SPECIAL, INDIRECT OR
|
||||
CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER
|
||||
TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
||||
PERFORMANCE OF THIS SOFTWARE.
|
||||
|
||||
|
||||
Copyright (c) 1996 Digital Equipment Corporation
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included
|
||||
in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL DIGITAL EQUIPMENT CORPORATION BE LIABLE FOR ANY CLAIM,
|
||||
DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
|
||||
OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR
|
||||
THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of the Digital Equipment
|
||||
Corporation shall not be used in advertising or otherwise to promote
|
||||
the sale, use or other dealings in this Software without prior written
|
||||
authorization from Digital Equipment Corporation.
|
||||
|
||||
|
||||
Copyright 1996, 1998 The Open Group
|
||||
|
||||
Permission to use, copy, modify, distribute, and sell this software and its
|
||||
documentation for any purpose is hereby granted without fee, provided that
|
||||
the above copyright notice appear in all copies and that both that
|
||||
copyright notice and this permission notice appear in supporting
|
||||
documentation.
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of The Open Group shall
|
||||
not be used in advertising or otherwise to promote the sale, use or
|
||||
other dealings in this Software without prior written authorization
|
||||
from The Open Group.
|
||||
|
||||
|
||||
Copyright 2004-2005 Sun Microsystems, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a
|
||||
copy of this software and associated documentation files (the "Software"),
|
||||
to deal in the Software without restriction, including without limitation
|
||||
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
and/or sell copies of the Software, and to permit persons to whom the
|
||||
Software is furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice (including the next
|
||||
paragraph) shall be included in all copies or substantial portions of the
|
||||
Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||
FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
DEALINGS IN THE SOFTWARE.
|
||||
|
||||
|
||||
Copyright (c) 1996 by Silicon Graphics Computer Systems, Inc.
|
||||
|
||||
Permission to use, copy, modify, and distribute this
|
||||
software and its documentation for any purpose and without
|
||||
fee is hereby granted, provided that the above copyright
|
||||
notice appear in all copies and that both that copyright
|
||||
notice and this permission notice appear in supporting
|
||||
documentation, and that the name of Silicon Graphics not be
|
||||
used in advertising or publicity pertaining to distribution
|
||||
of the software without specific prior written permission.
|
||||
Silicon Graphics makes no representation about the suitability
|
||||
of this software for any purpose. It is provided "as is"
|
||||
without any express or implied warranty.
|
||||
|
||||
SILICON GRAPHICS DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS
|
||||
SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
||||
AND FITNESS FOR A PARTICULAR PURPOSE. IN NO EVENT SHALL SILICON
|
||||
GRAPHICS BE LIABLE FOR ANY SPECIAL, INDIRECT OR CONSEQUENTIAL
|
||||
DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
|
||||
OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH
|
||||
THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
||||
|
||||
|
||||
Copyright (c) 1996 X Consortium
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE X CONSORTIUM BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of the X Consortium shall
|
||||
not be used in advertising or otherwise to promote the sale, use or
|
||||
other dealings in this Software without prior written authorization
|
||||
from the X Consortium.
|
||||
|
||||
|
||||
Copyright (C) 2004, 2006 Ævar Arnfjörð Bjarmason <avarab@gmail.com>
|
||||
|
||||
Permission to use, copy, modify, distribute, and sell this software and its
|
||||
documentation for any purpose is hereby granted without fee, provided that
|
||||
the above copyright notice appear in all copies and that both that
|
||||
copyright notice and this permission notice appear in supporting
|
||||
documentation.
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of a copyright holder shall
|
||||
not be used in advertising or otherwise to promote the sale, use or
|
||||
other dealings in this Software without prior written authorization of
|
||||
the copyright holder.
|
||||
|
||||
|
||||
Copyright (C) 1999, 2000 by Anton Zinoviev <anton@lml.bas.bg>
|
||||
|
||||
This software may be used, modified, copied, distributed, and sold,
|
||||
in both source and binary form provided that the above copyright
|
||||
and these terms are retained. Under no circumstances is the author
|
||||
responsible for the proper functioning of this software, nor does
|
||||
the author assume any responsibility for damages incurred with its
|
||||
use.
|
||||
|
||||
Permission is granted to anyone to use, distribute and modify
|
||||
this file in any way, provided that the above copyright notice
|
||||
is left intact and the author of the modification summarizes
|
||||
the changes in this header.
|
||||
|
||||
This file is distributed without any expressed or implied warranty.
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
# Vendored xkeyboard-config subset
|
||||
|
||||
- **Package**: xkeyboard-config 2.44
|
||||
- **Source**: https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config/-/archive/xkeyboard-config-2.44/xkeyboard-config-2.44.tar.gz
|
||||
- **sha256**: `35e34edeaf4e8da8d0696ff6b241ee11ddb1b8c6730bac7252d4d0a88ea5f05b`
|
||||
- **keysymdef.h**: xorgproto, copied from `/opt/homebrew/include/X11/keysymdef.h`
|
||||
- **License**: MIT/X11 (see COPYING)
|
||||
|
||||
Only the symbols files reachable from the generated layouts (tools/make-xkeyboard-config.py `TARGETS`) are vendored; regenerate with
|
||||
`python3 tools/make-xkeyboard-config.py fetch` then `... generate`.
|
||||
|
||||
Vendored symbols files:
|
||||
|
||||
- `symbols/de`
|
||||
- `symbols/es`
|
||||
- `symbols/fr`
|
||||
- `symbols/gb`
|
||||
- `symbols/kpdl`
|
||||
- `symbols/latin`
|
||||
- `symbols/level3`
|
||||
- `symbols/us`
|
||||
+2584
File diff suppressed because it is too large
Load Diff
+1232
File diff suppressed because it is too large
Load Diff
+250
@@ -0,0 +1,250 @@
|
||||
// Keyboard layouts for Spain.
|
||||
|
||||
// Modified for a real Spanish keyboard by Jon Tombs.
|
||||
default partial alphanumeric_keys
|
||||
xkb_symbols "basic" {
|
||||
|
||||
include "latin(type4)"
|
||||
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <TLDE> { [ masculine, ordfeminine, backslash, backslash ] };
|
||||
key <AE01> { [ 1, exclam, bar, exclamdown ] };
|
||||
key <AE03> { [ 3, periodcentered, numbersign, sterling ] };
|
||||
key <AE04> { [ 4, dollar, asciitilde, dollar ] };
|
||||
key <AE11> { [apostrophe, question, backslash, questiondown ] };
|
||||
key <AE12> { [exclamdown, questiondown, dead_cedilla, dead_ogonek] };
|
||||
|
||||
key <AD11> { [dead_grave, dead_circumflex, bracketleft, dead_abovering ] };
|
||||
key <AD12> { [ plus, asterisk, bracketright, dead_macron ] };
|
||||
|
||||
key <AC10> { [ ntilde, Ntilde, dead_tilde, dead_doubleacute ] };
|
||||
key <AC11> { [dead_acute, dead_diaeresis, braceleft, dead_caron ] };
|
||||
key <BKSL> { [ ccedilla, Ccedilla, braceright, dead_breve ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "winkeys" {
|
||||
|
||||
include "es(basic)"
|
||||
name[Group1]="Spanish (Windows)";
|
||||
include "eurosign(5)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "nodeadkeys" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Spanish (no dead keys)";
|
||||
|
||||
key <AE12> { [exclamdown, questiondown, cedilla, ogonek ] };
|
||||
key <AD11> { [ grave, asciicircum, bracketleft, degree ] };
|
||||
key <AD12> { [ plus, asterisk, bracketright, macron ] };
|
||||
key <AC07> { [ j, J, ezh, EZH ] };
|
||||
key <AC10> { [ ntilde, Ntilde, asciitilde, doubleacute ] };
|
||||
key <AC11> { [ acute, diaeresis, braceleft, caron ] };
|
||||
key <BKSL> { [ ccedilla, Ccedilla, braceright, breve ] };
|
||||
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
// Spanish Dvorak mapping (note R-H exchange)
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "dvorak" {
|
||||
|
||||
name[Group1]="Spanish (Dvorak)";
|
||||
|
||||
key <TLDE> {[ masculine, ordfeminine, backslash, degree ]};
|
||||
key <AE01> {[ 1, exclam, bar, onesuperior ]};
|
||||
key <AE02> {[ 2, quotedbl, at, twosuperior ]};
|
||||
key <AE03> {[ 3, periodcentered, numbersign, threesuperior ]};
|
||||
key <AE04> {[ 4, dollar, asciitilde, onequarter ]};
|
||||
key <AE05> {[ 5, percent, brokenbar, fiveeighths ]};
|
||||
key <AE06> {[ 6, ampersand, notsign, threequarters ]};
|
||||
key <AE07> {[ 7, slash, onehalf, seveneighths ]};
|
||||
key <AE08> {[ 8, parenleft, oneeighth, threeeighths ]};
|
||||
key <AE09> {[ 9, parenright, asciicircum ]};
|
||||
key <AE10> {[ 0, equal, grave, dead_doubleacute ]};
|
||||
key <AE11> {[ apostrophe, question, dead_macron, dead_ogonek ]};
|
||||
key <AE12> {[ exclamdown, questiondown, dead_breve, dead_abovedot ]};
|
||||
|
||||
key <AD01> {[ period, colon, less, guillemotleft ]};
|
||||
key <AD02> {[ comma, semicolon, greater, guillemotright ]};
|
||||
key <AD03> {[ ntilde, Ntilde, lstroke, Lstroke ]};
|
||||
key <AD04> {[ p, P, paragraph ]};
|
||||
key <AD05> {[ y, Y, yen ]};
|
||||
key <AD06> {[ f, F, tslash, Tslash ]};
|
||||
key <AD07> {[ g, G, dstroke, Dstroke ]};
|
||||
key <AD08> {[ c, C, cent, copyright ]};
|
||||
key <AD09> {[ h, H, hstroke, Hstroke ]};
|
||||
key <AD10> {[ l, L, sterling ]};
|
||||
key <AD11> {[ dead_grave, dead_circumflex, bracketleft, dead_caron ]};
|
||||
key <AD12> {[ plus, asterisk, bracketright, plusminus ]};
|
||||
|
||||
key <AC01> {[ a, A, ae, AE ]};
|
||||
key <AC02> {[ o, O, oslash, Oslash ]};
|
||||
key <AC03> {[ e, E, EuroSign ]};
|
||||
key <AC04> {[ u, U, aring, Aring ]};
|
||||
key <AC05> {[ i, I, oe, OE ]};
|
||||
key <AC06> {[ d, D, eth, ETH ]};
|
||||
key <AC07> {[ r, R, registered, trademark ]};
|
||||
key <AC08> {[ t, T, thorn, THORN ]};
|
||||
key <AC09> {[ n, N, eng, ENG ]};
|
||||
key <AC10> {[ s, S, ssharp, section ]};
|
||||
key <AC11> {[ dead_acute, dead_diaeresis, braceleft, dead_tilde ]};
|
||||
key <BKSL> {[ ccedilla, Ccedilla, braceright, dead_cedilla ]};
|
||||
|
||||
key <LSGT> {[ less, greater, guillemotleft, guillemotright ]};
|
||||
key <AB01> {[ minus, underscore, hyphen, macron ]};
|
||||
key <AB02> {[ q, Q, currency ]};
|
||||
key <AB03> {[ j, J ]};
|
||||
key <AB04> {[ k, K, kra ]};
|
||||
key <AB05> {[ x, X, multiply, division ]};
|
||||
key <AB06> {[ b, B ]};
|
||||
key <AB07> {[ m, M, mu ]};
|
||||
key <AB08> {[ w, W ]};
|
||||
key <AB09> {[ v, V ]};
|
||||
key <AB10> {[ z, Z ]};
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "cat" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Catalan (Spain, with middle-dot L)";
|
||||
|
||||
key <AC09> { [ l, L, 0x1000140, 0x100013F ] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "ast" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Asturian (Spain, with bottom-dot H and L)";
|
||||
|
||||
key <AC06> { [ h, H, 0x1001E25, 0x1001E24 ] };
|
||||
key <AC09> { [ l, L, 0x1001E37, 0x1001E36 ] };
|
||||
};
|
||||
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "olpc" {
|
||||
|
||||
// #HW-SPECIFIC
|
||||
|
||||
// http://wiki.laptop.org/go/OLPC_Spanish_Keyboard
|
||||
|
||||
include "us(basic)"
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <AE00> { [ masculine, ordfeminine ] };
|
||||
key <AE01> { [ 1, exclam, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, at ] };
|
||||
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||
key <AE06> { [ 6, ampersand, notsign ] };
|
||||
key <AE07> { [ 7, slash, backslash ] };
|
||||
key <AE08> { [ 8, parenleft ] };
|
||||
key <AE09> { [ 9, parenright ] };
|
||||
key <AE10> { [ 0, equal ] };
|
||||
key <AE11> { [ apostrophe, question ] };
|
||||
key <AE12> { [ exclamdown, questiondown ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign ] };
|
||||
key <AD11> { [ dead_acute, dead_diaeresis, acute, dead_abovering ] };
|
||||
key <AD12> { [ bracketleft, braceleft ] };
|
||||
|
||||
key <AC10> { [ ntilde, Ntilde ] };
|
||||
key <AC11> { [ plus, asterisk, dead_tilde ] };
|
||||
key <AC12> { [ bracketright, braceright, section ] };
|
||||
|
||||
key <AB08> { [ comma, semicolon ] };
|
||||
key <AB09> { [ period, colon ] };
|
||||
key <AB10> { [ minus, underscore ] };
|
||||
|
||||
key <I219> { [ less, greater, ISO_Next_Group ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "olpcm" {
|
||||
|
||||
// #HW-SPECIFIC
|
||||
|
||||
// Mechanical (non-membrane) OLPC Spanish keyboard layout.
|
||||
// See: http://wiki.laptop.org/go/OLPC_Spanish_Non-membrane_Keyboard
|
||||
|
||||
include "us(basic)"
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <AE00> { [ questiondown, exclamdown, backslash ] };
|
||||
key <AE01> { [ 1, exclam, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, at ] };
|
||||
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||
key <AE04> { [ 4, dollar, asciitilde, dead_tilde ] };
|
||||
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||
key <AE06> { [ 6, ampersand, notsign ] };
|
||||
key <AE07> { [ 7, slash, backslash ] }; // no '\' label on olpcm, leave for compatibility
|
||||
key <AE08> { [ 8, parenleft, masculine ] };
|
||||
key <AE09> { [ 9, parenright, ordfeminine ] };
|
||||
key <AE10> { [ 0, equal ] };
|
||||
key <AE11> { [ apostrophe, question ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign ] };
|
||||
key <AD11> { [ dead_acute, dead_diaeresis, dead_abovering, acute ] };
|
||||
key <AD12> { [ plus, asterisk ] };
|
||||
|
||||
key <AC10> { [ ntilde, Ntilde ] };
|
||||
// no AC11 or AC12 on olpcm
|
||||
|
||||
key <AB08> { [ comma, semicolon ] };
|
||||
key <AB09> { [ period, colon ] };
|
||||
key <AB10> { [ minus, underscore ] };
|
||||
|
||||
key <AA02> { [ less, greater ] };
|
||||
key <AA06> { [ bracketleft, braceleft, ccedilla, Ccedilla ] };
|
||||
key <AA07> { [ bracketright, braceright ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "deadtilde" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Spanish (dead tilde)";
|
||||
|
||||
key <AE04> { [ 4, dollar, dead_tilde, dollar ] };
|
||||
key <AC10> { [ ntilde, Ntilde, asciitilde, dead_doubleacute ] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "olpc2" {
|
||||
// #HW-SPECIFIC
|
||||
|
||||
// Modified variant of US International layout, specifically for Peru
|
||||
// Contact: Sayamindu Dasgupta <sayamindu@laptop.org>
|
||||
|
||||
include "us(olpc)"
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <AE03> { [ 3, numbersign, dead_grave, dead_grave] }; // combining grave
|
||||
key <I236> { [ XF86Start ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
// EXTRAS:
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "sun_type6" {
|
||||
include "sun_vndr/es(sun_type6)"
|
||||
};
|
||||
+1404
File diff suppressed because it is too large
Load Diff
+249
@@ -0,0 +1,249 @@
|
||||
// Keyboard layouts for Great Britain.
|
||||
|
||||
default partial alphanumeric_keys
|
||||
xkb_symbols "basic" {
|
||||
|
||||
// The basic UK layout, also known as the IBM 166 layout,
|
||||
// but with the useless brokenbar pushed two levels up.
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, bar, brokenbar ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "intl" {
|
||||
|
||||
// A UK layout but with five accents made into dead keys:
|
||||
// grave, diaeresis, circumflex, acute, and tilde.
|
||||
// By Phil Jones <philjones1 at blueyonder.co.uk>.
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK, intl., with dead keys)";
|
||||
|
||||
key <TLDE> { [ dead_grave, notsign, bar, bar ] };
|
||||
key <AE02> { [ 2, dead_diaeresis, twosuperior, onehalf ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
key <AE06> { [ 6, dead_circumflex, threequarters, onesixth ] };
|
||||
|
||||
key <AC11> { [ dead_acute, at, apostrophe, bar ] };
|
||||
key <BKSL> { [ numbersign, dead_tilde, bar, bar ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, bar, bar ] };
|
||||
key <AB08> { [ comma, less, ccedilla, Ccedilla ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "extd" {
|
||||
// Clone of the Microsoft "United Kingdom Extended" layout, which
|
||||
// includes dead keys for: grave; diaeresis; circumflex; tilde; and
|
||||
// accute. It also enables direct access to accute characters using
|
||||
// the Multi_key (Alt Gr).
|
||||
//
|
||||
// Taken from...
|
||||
// "Windows Keyboard Layouts"
|
||||
// https://docs.microsoft.com/en-gb/globalization/windows-keyboard-layouts#U
|
||||
//
|
||||
// -- Jonathan Miles <jon@cybah.co.uk>
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK, extended, Windows)";
|
||||
|
||||
key <TLDE> { [ dead_grave, notsign, brokenbar, NoSymbol ] };
|
||||
key <AE02> { [ 2, quotedbl, dead_diaeresis, onehalf ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
key <AE06> { [ 6, asciicircum, dead_circumflex, NoSymbol ] };
|
||||
|
||||
key <AD02> { [ w, W, wacute, Wacute ] };
|
||||
key <AD03> { [ e, E, eacute, Eacute ] };
|
||||
key <AD06> { [ y, Y, yacute, Yacute ] };
|
||||
key <AD07> { [ u, U, uacute, Uacute ] };
|
||||
key <AD08> { [ i, I, iacute, Iacute ] };
|
||||
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||
key <AD12> { [ bracketright, braceright, NoSymbol, bar ] };
|
||||
|
||||
key <AC01> { [ a, A, aacute, Aacute ] };
|
||||
key <AC11> { [ apostrophe, at, dead_acute, grave ] };
|
||||
key <BKSL> { [ numbersign, asciitilde, dead_tilde, backslash ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, NoSymbol, NoSymbol ] };
|
||||
key <AB03> { [ c, C, ccedilla, Ccedilla ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
// Describe the differences between the US Colemak layout
|
||||
// and a UK variant. By Andy Buckley (andy@insectnation.org)
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "colemak" {
|
||||
include "us(colemak)"
|
||||
|
||||
name[Group1]="English (UK, Colemak)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||
};
|
||||
|
||||
// Colemak-DH (ISO) layout, UK Variant, https://colemakmods.github.io/mod-dh/
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "colemak_dh" {
|
||||
include "us(colemak_dh)"
|
||||
|
||||
name[Group1]="English (UK, Colemak-DH)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||
|
||||
key <AB05> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||
};
|
||||
|
||||
|
||||
// Dvorak (UK) keymap (by odaen) allowing the usage of
|
||||
// the £ and ? key and swapping the @ and " keys.
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "dvorak" {
|
||||
include "us(dvorak-alt-intl)"
|
||||
|
||||
name[Group1]="English (UK, Dvorak)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, NoSymbol ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, NoSymbol ] };
|
||||
key <AD01> { [ apostrophe, at ] };
|
||||
key <BKSL> { [ numbersign, asciitilde ] };
|
||||
key <LSGT> { [ backslash, bar ] };
|
||||
};
|
||||
|
||||
// Dvorak letter positions, but punctuation all in the normal UK positions.
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "dvorakukp" {
|
||||
include "gb(dvorak)"
|
||||
|
||||
name[Group1]="English (UK, Dvorak, with UK punctuation)";
|
||||
|
||||
key <AE11> { [ minus, underscore ] };
|
||||
key <AE12> { [ equal, plus ] };
|
||||
key <AD11> { [ bracketleft, braceleft ] };
|
||||
key <AD12> { [ bracketright, braceright ] };
|
||||
key <AD01> { [ slash, question ] };
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "mac" {
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]= "English (UK, Macintosh)";
|
||||
|
||||
key <TLDE> { [ section, plusminus ] };
|
||||
key <AE02> { [ 2, at, EuroSign ] };
|
||||
key <AE03> { [ 3, sterling, numbersign ] };
|
||||
key <LSGT> { [ grave, asciitilde ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
include "level3(enter_switch)"
|
||||
};
|
||||
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "mac_intl" {
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK, Macintosh, intl.)";
|
||||
|
||||
key <TLDE> { [ section, plusminus, notsign, notsign ] }; //dead_grave
|
||||
key <AE02> { [ 2, at, EuroSign, onehalf ] };
|
||||
key <AE03> { [ 3, sterling, twosuperior, onethird ] };
|
||||
key <AE04> { [ 4, dollar, threesuperior, onequarter ] };
|
||||
key <AE06> { [ 6, dead_circumflex, NoSymbol, onesixth ] };
|
||||
key <AD09> { [ o, O, oe, OE ] };
|
||||
|
||||
key <AC11> { [ dead_acute, dead_diaeresis, dead_diaeresis, bar ] }; //dead_doubleacute
|
||||
key <BKSL> { [ backslash, bar, numbersign, bar ] };
|
||||
|
||||
key <LSGT> { [ dead_grave, dead_tilde, brokenbar, bar ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "pl" {
|
||||
|
||||
// Polish accented letters on upper levels of corresponding base letters.
|
||||
// Idea from Wawrzyniec Niewodniczański, adapted by Aleksander Kowalski.
|
||||
|
||||
include "gb(basic)"
|
||||
|
||||
name[Group1]="Polish (British keyboard)";
|
||||
|
||||
key <AD03> { [ e, E, eogonek, Eogonek ] };
|
||||
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||
|
||||
key <AC01> { [ a, A, aogonek, Aogonek ] };
|
||||
key <AC02> { [ s, S, sacute, Sacute ] };
|
||||
|
||||
key <AB01> { [ z, Z, zabovedot, Zabovedot ] };
|
||||
key <AB02> { [ x, X, zacute, Zacute ] };
|
||||
key <AB03> { [ c, C, cacute, Cacute ] };
|
||||
key <AB06> { [ n, N, nacute, Nacute ] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "gla" {
|
||||
|
||||
// Grave-accented letters on the upper levels of the relevant vowels.
|
||||
|
||||
include "gb(basic)"
|
||||
|
||||
name[Group1]="Scottish Gaelic";
|
||||
|
||||
key <AD03> { [ e, E, egrave, Egrave ] };
|
||||
key <AD07> { [ u, U, ugrave, Ugrave ] };
|
||||
key <AD08> { [ i, I, igrave, Igrave ] };
|
||||
key <AD09> { [ o, O, ograve, Ograve ] };
|
||||
|
||||
key <AC01> { [ a, A, agrave, Agrave ] };
|
||||
};
|
||||
|
||||
// EXTRAS:
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "sun_type6" {
|
||||
include "sun_vndr/gb(sun_type6)"
|
||||
};
|
||||
+102
@@ -0,0 +1,102 @@
|
||||
// The <KPDL> key is a mess.
|
||||
// It was probably originally meant to be a decimal separator.
|
||||
// Except since it was declared by USA people it didn't use the original
|
||||
// SI separator "," but a "." (since then the USA managed to f-up the SI
|
||||
// by making "." an accepted alternative, but standards still use "," as
|
||||
// default)
|
||||
// As a result users of SI-abiding countries expect either a "." or a ","
|
||||
// or a "decimal_separator" which may or may not be translated in one of the
|
||||
// above depending on applications.
|
||||
// It's not possible to define a default per-country since user expectations
|
||||
// depend on the conflicting choices of their most-used applications,
|
||||
// operating system, etc. Therefore it needs to be a configuration setting
|
||||
// Copyright © 2007 Nicolas Mailhot <nicolas.mailhot @ laposte.net>
|
||||
|
||||
|
||||
// Legacy <KPDL> #1
|
||||
// This assumes KP_Decimal will be translated in a dot
|
||||
partial keypad_keys
|
||||
xkb_symbols "dot" {
|
||||
|
||||
key.type[Group1]="KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, KP_Decimal ] }; // <delete> <separator>
|
||||
};
|
||||
|
||||
|
||||
// Legacy <KPDL> #2
|
||||
// This assumes KP_Separator will be translated in a comma
|
||||
partial keypad_keys
|
||||
xkb_symbols "comma" {
|
||||
|
||||
key.type[Group1]="KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, KP_Separator ] }; // <delete> <separator>
|
||||
};
|
||||
|
||||
|
||||
// Period <KPDL>, usual keyboard serigraphy in most countries
|
||||
partial keypad_keys
|
||||
xkb_symbols "dotoss" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, period, comma, 0x100202F ] }; // <delete> . , ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Period <KPDL>, usual keyboard serigraphy in most countries, latin-9 restriction
|
||||
partial keypad_keys
|
||||
xkb_symbols "dotoss_latin9" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, period, comma, nobreakspace ] }; // <delete> . , ⍽ (no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Comma <KPDL>, what most non anglo-saxon people consider the real separator
|
||||
partial keypad_keys
|
||||
xkb_symbols "commaoss" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, comma, period, 0x100202F ] }; // <delete> , . ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Momayyez <KPDL>: Bahrain, Iran, Iraq, Kuwait, Oman, Qatar, Saudi Arabia, Syria, UAE
|
||||
partial keypad_keys
|
||||
xkb_symbols "momayyezoss" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, 0x100066B, comma, 0x100202F ] }; // <delete> ? , ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Abstracted <KPDL>, pray everything will work out (it usually does not)
|
||||
partial keypad_keys
|
||||
xkb_symbols "kposs" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, KP_Decimal, KP_Separator, 0x100202F ] }; // <delete> ? ? ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
// Spreadsheets may be configured to use the dot as decimal
|
||||
// punctuation, comma as a thousands separator and then semi-colon as
|
||||
// the list separator. Of these, dot and semi-colon is most important
|
||||
// when entering data by the keyboard; the comma can then be inferred
|
||||
// and added to the presentation afterwards. Using semi-colon as a
|
||||
// general separator may in fact be preferred to avoid ambiguities
|
||||
// in data files. Most times a decimal separator is hard-coded, it
|
||||
// seems to be period, probably since this is the syntax used in
|
||||
// (most) programming languages.
|
||||
partial keypad_keys
|
||||
xkb_symbols "semi" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ NoSymbol, NoSymbol, semicolon ] };
|
||||
};
|
||||
+255
@@ -0,0 +1,255 @@
|
||||
// Common Latin alphabet layout
|
||||
|
||||
default partial
|
||||
xkb_symbols "basic" {
|
||||
|
||||
key <AE01> { [ 1, exclam, onesuperior, exclamdown ] };
|
||||
key <AE02> { [ 2, at, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, numbersign, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, onequarter, dollar ] };
|
||||
key <AE05> { [ 5, percent, onehalf, threeeighths ] };
|
||||
key <AE06> { [ 6, asciicircum, threequarters, fiveeighths ] };
|
||||
key <AE07> { [ 7, ampersand, braceleft, seveneighths ] };
|
||||
key <AE08> { [ 8, asterisk, bracketleft, trademark ] };
|
||||
key <AE09> { [ 9, parenleft, bracketright, plusminus ] };
|
||||
key <AE10> { [ 0, parenright, braceright, degree ] };
|
||||
key <AE11> { [ minus, underscore, backslash, questiondown ] };
|
||||
key <AE12> { [ equal, plus, dead_cedilla, dead_ogonek ] };
|
||||
|
||||
key <AD01> { [ q, Q, at, Greek_OMEGA ] };
|
||||
key <AD02> { [ w, W, U017F, section ] };
|
||||
key <AD03> { [ e, E, e, E ] };
|
||||
key <AD04> { [ r, R, paragraph, registered ] };
|
||||
key <AD05> { [ t, T, tslash, Tslash ] };
|
||||
key <AD06> { [ y, Y, leftarrow, yen ] };
|
||||
key <AD07> { [ u, U, downarrow, uparrow ] };
|
||||
key <AD08> { [ i, I, rightarrow, idotless ] };
|
||||
key <AD09> { [ o, O, oslash, Oslash ] };
|
||||
key <AD10> { [ p, P, thorn, THORN ] };
|
||||
key <AD11> { [bracketleft, braceleft, dead_diaeresis, dead_abovering ] };
|
||||
key <AD12> { [bracketright, braceright, dead_tilde, dead_macron ] };
|
||||
|
||||
key <AC01> { [ a, A, ae, AE ] };
|
||||
key <AC02> { [ s, S, ssharp, U1E9E ] };
|
||||
key <AC03> { [ d, D, eth, ETH ] };
|
||||
key <AC04> { [ f, F, dstroke, ordfeminine ] };
|
||||
key <AC05> { [ g, G, eng, ENG ] };
|
||||
key <AC06> { [ h, H, hstroke, Hstroke ] };
|
||||
key <AC07> { [ j, J, dead_hook, dead_horn ] };
|
||||
key <AC08> { [ k, K, kra, ampersand ] };
|
||||
key <AC09> { [ l, L, lstroke, Lstroke ] };
|
||||
key <AC10> { [ semicolon, colon, dead_acute, dead_doubleacute ] };
|
||||
key <AC11> { [apostrophe, quotedbl, dead_circumflex, dead_caron ] };
|
||||
key <TLDE> { [ grave, asciitilde, notsign, notsign ] };
|
||||
|
||||
key <BKSL> { [ backslash, bar, dead_grave, dead_breve ] };
|
||||
key <AB01> { [ z, Z, guillemotleft, less ] };
|
||||
key <AB02> { [ x, X, guillemotright, greater ] };
|
||||
key <AB03> { [ c, C, cent, copyright ] };
|
||||
key <AB04> { [ v, V, doublelowquotemark, singlelowquotemark ] };
|
||||
key <AB05> { [ b, B, leftdoublequotemark, leftsinglequotemark ] };
|
||||
key <AB06> { [ n, N, rightdoublequotemark, rightsinglequotemark ] };
|
||||
key <AB07> { [ m, M, mu, masculine ] };
|
||||
key <AB08> { [ comma, less, U2022, multiply ] }; // bullet
|
||||
key <AB09> { [ period, greater, periodcentered, division ] };
|
||||
key <AB10> { [ slash, question, dead_belowdot, dead_abovedot ] };
|
||||
};
|
||||
|
||||
// Northern Europe ( Danish, Finnish, Norwegian, Swedish) common layout
|
||||
|
||||
partial
|
||||
xkb_symbols "type2" {
|
||||
|
||||
include "latin"
|
||||
|
||||
key <AE01> { [ 1, exclam, exclamdown, onesuperior ] };
|
||||
key <AE02> { [ 2, quotedbl, at, twosuperior ] };
|
||||
key <AE03> { [ 3, numbersign, sterling, threesuperior] };
|
||||
key <AE04> { [ 4, currency, dollar, onequarter ] };
|
||||
key <AE05> { [ 5, percent, onehalf, cent ] };
|
||||
key <AE06> { [ 6, ampersand, yen, fiveeighths ] };
|
||||
key <AE07> { [ 7, slash, braceleft, division ] };
|
||||
key <AE08> { [ 8, parenleft, bracketleft, guillemotleft] };
|
||||
key <AE09> { [ 9, parenright, bracketright, guillemotright] };
|
||||
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||
key <AD04> { [ r, R, registered, registered ] };
|
||||
key <AD05> { [ t, T, thorn, THORN ] };
|
||||
key <AD09> { [ o, O, oe, OE ] };
|
||||
key <AD11> { [ aring, Aring, dead_diaeresis, dead_abovering ] };
|
||||
key <AD12> { [dead_diaeresis, dead_circumflex, dead_tilde, dead_caron ] };
|
||||
|
||||
key <AC01> { [ a, A, ordfeminine, masculine ] };
|
||||
|
||||
key <AB03> { [ c, C, copyright, copyright ] };
|
||||
key <AB08> { [ comma, semicolon, dead_cedilla, dead_ogonek ] };
|
||||
key <AB09> { [ period, colon, periodcentered, dead_abovedot ] };
|
||||
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||
};
|
||||
|
||||
// Slavic Latin ( Albanian, Croatian, Polish, Slovene, Yugoslav)
|
||||
// common layout
|
||||
|
||||
partial
|
||||
xkb_symbols "type3" {
|
||||
|
||||
include "latin"
|
||||
|
||||
key <AD01> { [ q, Q, backslash, Greek_OMEGA ] };
|
||||
key <AD02> { [ w, W, bar, section ] };
|
||||
key <AD06> { [ z, Z, leftarrow, yen ] };
|
||||
|
||||
key <AC04> { [ f, F, bracketleft, ordfeminine ] };
|
||||
key <AC05> { [ g, G, bracketright, ENG ] };
|
||||
key <AC08> { [ k, K, lstroke, ampersand ] };
|
||||
|
||||
key <AB01> { [ y, Y, guillemotleft, less ] };
|
||||
key <AB04> { [ v, V, at, grave ] };
|
||||
key <AB05> { [ b, B, braceleft, apostrophe ] };
|
||||
key <AB06> { [ n, N, braceright, acute ] };
|
||||
key <AB07> { [ m, M, section, masculine ] };
|
||||
key <AB08> { [ comma, semicolon, less, multiply ] };
|
||||
key <AB09> { [ period, colon, greater, division ] };
|
||||
};
|
||||
|
||||
// Another common Latin layout
|
||||
// (German, Estonian, Spanish, Icelandic, Italian, Latin American, Portuguese)
|
||||
|
||||
partial
|
||||
xkb_symbols "type4" {
|
||||
|
||||
include "latin"
|
||||
|
||||
key <AE02> { [ 2, quotedbl, at, oneeighth ] };
|
||||
key <AE06> { [ 6, ampersand, notsign, fiveeighths ] };
|
||||
key <AE07> { [ 7, slash, braceleft, seveneighths ] };
|
||||
key <AE08> { [ 8, parenleft, bracketleft, trademark ] };
|
||||
key <AE09> { [ 9, parenright, bracketright, plusminus ] };
|
||||
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||
|
||||
key <AB08> { [ comma, semicolon, U2022, multiply ] }; // bullet
|
||||
key <AB09> { [ period, colon, periodcentered, division ] };
|
||||
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "nodeadkeys" {
|
||||
|
||||
key <AE12> { [ equal, plus, cedilla, ogonek ] };
|
||||
key <AD11> { [bracketleft, braceleft, diaeresis, degree ] };
|
||||
key <AD12> { [bracketright, braceright, asciitilde, macron ] };
|
||||
key <AC07> { [ j, J, ezh, EZH ] };
|
||||
key <AC10> { [ semicolon, colon, acute, doubleacute ] };
|
||||
key <AC11> { [apostrophe, quotedbl, asciicircum, caron ] };
|
||||
key <BKSL> { [ backslash, bar, grave, breve ] };
|
||||
key <AB10> { [ slash, question, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "type2_nodeadkeys" {
|
||||
|
||||
include "latin(nodeadkeys)"
|
||||
|
||||
key <AD11> { [ aring, Aring, diaeresis, degree ] };
|
||||
key <AD12> { [ diaeresis, asciicircum, asciitilde, caron ] };
|
||||
key <AB08> { [ comma, semicolon, cedilla, ogonek ] };
|
||||
key <AB09> { [ period, colon, periodcentered, abovedot ] };
|
||||
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "type3_nodeadkeys" {
|
||||
|
||||
include "latin(nodeadkeys)"
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "type4_nodeadkeys" {
|
||||
|
||||
include "latin(nodeadkeys)"
|
||||
|
||||
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
// Added 2008.03.05 by Marcin Woliński
|
||||
// See http://marcinwolinski.pl/keyboard/ for a description.
|
||||
// Used by pl(intl)
|
||||
//
|
||||
// ┌─────┐
|
||||
// │ 2 4 │ 2 = Shift, 4 = Level3 + Shift
|
||||
// │ 1 3 │ 1 = Normal, 3 = Level3
|
||||
// └─────┘
|
||||
// ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┲━━━━━━━━━┓
|
||||
// │ ~ ~ │ ! ' │ @ " │ # ˝ │ $ ¸ │ % ˇ │ ^ ^ │ & ˘ │ * ̇ │ ( ̣ │ ) ° │ _ ¯ │ + ˛ ┃ ⌫ Back- ┃
|
||||
// │ ` ` │ 1 ¡ │ 2 © │ 3 • │ 4 § │ 5 € │ 6 ¢ │ 7 − │ 8 × │ 9 ÷ │ 0 ° │ - – │ = — ┃ space ┃
|
||||
// ┢━━━━━┷━┱───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┺━┳━━━━━━━┫
|
||||
// ┃ ┃ Q │ W │ E │ R │ T │ Y │ U │ I │ O │ P │ { « │ } » ┃ Enter ┃
|
||||
// ┃Tab ↹ ┃ q │ w │ e │ r │ t │ y │ u │ i │ o │ p │ [ ‹ │ ] › ┃ ⏎ ┃
|
||||
// ┣━━━━━━━┻┱────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┺┓ ┃
|
||||
// ┃ ┃ A │ S │ D │ F │ G │ H │ J │ K │ L │ : “ │ " ” │ | ¶ ┃ ┃
|
||||
// ┃Caps ⇬ ┃ a │ s │ d │ f │ g │ h │ j │ k │ l │ ; ‘ │ ' ’ │ \ ┃ ┃
|
||||
// ┣━━━━━━━━┹────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┲┷━━━━━┻━━━━━━┫
|
||||
// ┃ │ Z │ X │ C │ V │ B │ N │ M │ < „ │ > · │ ? ¿ ┃ ┃
|
||||
// ┃Shift ⇧ │ z │ x │ c │ v │ b │ n │ m │ , ‚ │ . … │ / ⁄ ┃Shift ⇧ ┃
|
||||
// ┣━━━━━━━┳━━━━━┷━┳━━━┷━━━┱─┴─────┴─────┴─────┴─────┴─────┴───┲━┷━━━━━╈━━━━━┻━┳━━━━━━━┳━━━┛
|
||||
// ┃ ┃ ┃ ┃ ␣ ⍽ ┃ ┃ ┃ ┃
|
||||
// ┃Ctrl ┃Meta ┃Alt ┃ ␣ Space ⍽ ┃AltGr ⇮┃Menu ┃Ctrl ┃
|
||||
// ┗━━━━━━━┻━━━━━━━┻━━━━━━━┹───────────────────────────────────┺━━━━━━━┻━━━━━━━┻━━━━━━━┛
|
||||
|
||||
partial
|
||||
xkb_symbols "intl" {
|
||||
|
||||
key <TLDE> { [ grave, asciitilde, dead_grave, dead_tilde ] };
|
||||
key <AE01> { [ 1, exclam, exclamdown, dead_acute ] };
|
||||
key <AE02> { [ 2, at, copyright, dead_diaeresis ] };
|
||||
key <AE03> { [ 3, numbersign, U2022, dead_doubleacute ] }; // U+2022 is bullet (the name bullet does not work)
|
||||
key <AE04> { [ 4, dollar, section, dead_cedilla ] };
|
||||
key <AE05> { [ 5, percent, EuroSign, dead_caron ] };
|
||||
key <AE06> { [ 6, asciicircum, cent, dead_circumflex ] };
|
||||
key <AE07> { [ 7, ampersand, U2212, dead_breve ] }; // U+2212 is MINUS SIGN
|
||||
key <AE08> { [ 8, asterisk, multiply, dead_abovedot ] };
|
||||
key <AE09> { [ 9, parenleft, division, dead_belowdot ] };
|
||||
key <AE10> { [ 0, parenright, degree, dead_abovering ] };
|
||||
key <AE11> { [ minus, underscore, endash, dead_macron ] };
|
||||
key <AE12> { [ equal, plus, emdash, dead_ogonek ] };
|
||||
|
||||
key <AD01> { [ q, Q ] };
|
||||
key <AD02> { [ w, W ] };
|
||||
key <AD03> { [ e, E ] };
|
||||
key <AD04> { [ r, R ] };
|
||||
key <AD05> { [ t, T ] };
|
||||
key <AD06> { [ y, Y ] };
|
||||
key <AD07> { [ u, U ] };
|
||||
key <AD08> { [ i, I ] };
|
||||
key <AD09> { [ o, O ] };
|
||||
key <AD10> { [ p, P ] };
|
||||
key <AD11> { [bracketleft, braceleft, U2039, guillemotleft ] };
|
||||
key <AD12> { [bracketright, braceright, U203A, guillemotright ] };
|
||||
|
||||
key <AC01> { [ a, A ] };
|
||||
key <AC02> { [ s, S ] };
|
||||
key <AC03> { [ d, D ] };
|
||||
key <AC04> { [ f, F ] };
|
||||
key <AC05> { [ g, G ] };
|
||||
key <AC06> { [ h, H ] };
|
||||
key <AC07> { [ j, J ] };
|
||||
key <AC08> { [ k, K ] };
|
||||
key <AC09> { [ l, L ] };
|
||||
key <AC10> { [ semicolon, colon, leftsinglequotemark, leftdoublequotemark ] };
|
||||
key <AC11> { [apostrophe, quotedbl, rightsinglequotemark, rightdoublequotemark ] };
|
||||
|
||||
key <BKSL> { [ backslash, bar, NoSymbol, paragraph ] };
|
||||
key <AB01> { [ z, Z ] };
|
||||
key <AB02> { [ x, X ] };
|
||||
key <AB03> { [ c, C ] };
|
||||
key <AB04> { [ v, V ] };
|
||||
key <AB05> { [ b, B ] };
|
||||
key <AB06> { [ n, N ] };
|
||||
key <AB07> { [ m, M ] };
|
||||
key <AB08> { [ comma, less, singlelowquotemark, doublelowquotemark ] };
|
||||
key <AB09> { [ period, greater, ellipsis, periodcentered ] };
|
||||
key <AB10> { [ slash, question, U2044, questiondown ] }; // U+2044 is FRACTION SLASH
|
||||
};
|
||||
+156
@@ -0,0 +1,156 @@
|
||||
// These variants assign ISO_Level3_Shift to various keys
|
||||
// so that levels 3 and 4 can be reached.
|
||||
|
||||
// The default behaviour:
|
||||
// the right Alt key (AltGr) chooses the third symbol engraved on a key.
|
||||
default partial modifier_keys
|
||||
xkb_symbols "ralt_switch" {
|
||||
key <RALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The right Alt key never chooses the third level.
|
||||
// This option attempts to undo the effect of a layout's inclusion of
|
||||
// 'ralt_switch'. You may want to also select another level3 option
|
||||
// to map the level3 shift to some other key.
|
||||
partial modifier_keys
|
||||
xkb_symbols "ralt_alt" {
|
||||
key <RALT> {[ Alt_R, Meta_R ], type[group1]="TWO_LEVEL" };
|
||||
modifier_map Mod1 { <RALT> };
|
||||
};
|
||||
|
||||
// The right Alt key (while pressed) chooses the third shift level,
|
||||
// and Compose is mapped to its second level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "ralt_switch_multikey" {
|
||||
key <RALT> {[ ISO_Level3_Shift, Multi_key ], type[group1]="TWO_LEVEL" };
|
||||
};
|
||||
|
||||
// Either Alt key (while pressed) chooses the third shift level.
|
||||
// (To be used mostly to imitate Mac OS functionality.)
|
||||
partial modifier_keys
|
||||
xkb_symbols "alt_switch" {
|
||||
include "level3(lalt_switch)"
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
// The left Alt key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lalt_switch" {
|
||||
key <LALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The right Ctrl key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "switch" {
|
||||
key <RCTL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Menu key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "menu_switch" {
|
||||
key <MENU> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// Either Win key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "win_switch" {
|
||||
include "level3(lwin_switch)"
|
||||
include "level3(rwin_switch)"
|
||||
};
|
||||
|
||||
// The left Win key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lwin_switch" {
|
||||
key <LWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The right Win key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "rwin_switch" {
|
||||
key <RWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Enter key on the kepypad (while pressed) chooses the third shift level.
|
||||
// (This is especially useful for Mac laptops which miss the right Alt key.)
|
||||
partial modifier_keys
|
||||
xkb_symbols "enter_switch" {
|
||||
key <KPEN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The CapsLock key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "caps_switch" {
|
||||
key <CAPS> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The CapsLock key (while pressed) chooses the third shift level and
|
||||
// Ctrl + CapsLock has the original CapsLock function.
|
||||
// The 2023 DIN standard for German keyboards recommends it as an option:
|
||||
// - https://de.wikipedia.org/wiki/E1_(Tastaturbelegung)#Feststelltaste/Umschaltsperre
|
||||
// - https://en.wikipedia.org/wiki/Caps_Lock#Abolition
|
||||
partial modifier_keys
|
||||
xkb_symbols "caps_switch_capslock_with_ctrl" {
|
||||
virtual_modifiers LevelThree;
|
||||
|
||||
key <CAPS> {
|
||||
type[Group1] = "PC_CONTROL_LEVEL2",
|
||||
symbols[Group1] = [ ISO_Level3_Shift, Caps_Lock ],
|
||||
// Explicit actions are preferred over modMap None/Mod5 { Caps_Lock }
|
||||
// because they have no side effect
|
||||
actions[Group1] = [ SetMods(modifiers = LevelThree), LockMods(modifiers = Lock) ]
|
||||
};
|
||||
};
|
||||
|
||||
// The Backslash key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "bksl_switch" {
|
||||
key <BKSL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The AC11 key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "ac11_switch" {
|
||||
key <AC11> {[ ISO_Level3_Shift ], type[Group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Less/Greater key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lsgt_switch" {
|
||||
key <LSGT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The CapsLock key (while pressed) chooses the third shift level,
|
||||
// and latches when pressed together with another third-level chooser.
|
||||
partial modifier_keys
|
||||
xkb_symbols "caps_switch_latch" {
|
||||
key <CAPS> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||
type[group1]="THREE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Backslash key (while pressed) chooses the third shift level,
|
||||
// and latches when pressed together with another third-level chooser.
|
||||
partial modifier_keys
|
||||
xkb_symbols "bksl_switch_latch" {
|
||||
key <BKSL> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||
type[group1]="THREE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Less/Greater key (while pressed) chooses the third shift level,
|
||||
// and latches when pressed together with another third-level chooser.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lsgt_switch_latch" {
|
||||
key <LSGT> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||
type[group1]="THREE_LEVEL" };
|
||||
};
|
||||
|
||||
// Top-row digit key 4 chooses third shift level when pressed alone.
|
||||
partial modifier_keys
|
||||
xkb_symbols "4_switch_isolated" {
|
||||
override key <AE04> {[ ISO_Level3_Shift ]};
|
||||
};
|
||||
|
||||
// Top-row digit key 9 chooses third shift level when pressed alone.
|
||||
partial modifier_keys
|
||||
xkb_symbols "9_switch_isolated" {
|
||||
override key <AE09> {[ ISO_Level3_Shift ]};
|
||||
};
|
||||
+2238
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,153 @@
|
||||
//! xkeyboard-config — keyboard layouts, compiled from the X11 xkeyboard-config database
|
||||
//! into native Zig. It turns a physical key (a USB HID usage, as the input module delivers)
|
||||
//! plus a modifier state into a **keysym** and, when the key produces one, a **character**
|
||||
//! (a Unicode scalar). This is the piece that lets a `KeyEvent.keycode` become a
|
||||
//! `KeyEvent.character`, without shipping an X11 runtime.
|
||||
//!
|
||||
//! The layout tables in `generated/layouts.zig` are produced by
|
||||
//! `tools/make-xkeyboard-config.py` (see ./README.md to regenerate). Those tables are
|
||||
//! deliberately pure data — each key carries its up-to-four levels and an XKB *type*. The
|
||||
//! type -> level selection semantics (which modifier picks which level) live here, so the
|
||||
//! data and the policy are separable.
|
||||
//!
|
||||
//! Scope (documented in README.md): group 1 only, no dead-key/compose composition (a dead
|
||||
//! key returns its keysym with no character), and a curated set of key types. Layouts:
|
||||
//! us, gb, de, fr, es, dvorak.
|
||||
//!
|
||||
//! Upstream xkeyboard-config and keysymdef.h are MIT/X11 licensed; see vendor/COPYING and
|
||||
//! vendor/PROVENANCE.md.
|
||||
|
||||
const std = @import("std");
|
||||
const generated = @import("layouts");
|
||||
|
||||
pub const Level = generated.Level;
|
||||
pub const KeyType = generated.KeyType;
|
||||
pub const Key = generated.Key;
|
||||
pub const Layout = generated.Layout;
|
||||
|
||||
/// The generated layouts, by name — as pointers, so they share identity with `all` and
|
||||
/// `byName` (and match the `*const Layout` that `map` takes).
|
||||
pub const us: *const Layout = &generated.us;
|
||||
pub const gb: *const Layout = &generated.gb;
|
||||
pub const de: *const Layout = &generated.de;
|
||||
pub const fr: *const Layout = &generated.fr;
|
||||
pub const es: *const Layout = &generated.es;
|
||||
pub const dvorak: *const Layout = &generated.dvorak;
|
||||
|
||||
/// Every generated layout, for enumeration (e.g. a settings UI).
|
||||
pub const all = generated.all;
|
||||
|
||||
/// The modifier state that selects a key's level. `level3` is AltGr (ISO Level3 Shift);
|
||||
/// `control` is accepted for completeness but does not affect level selection here.
|
||||
pub const Modifiers = struct {
|
||||
shift: bool = false,
|
||||
caps_lock: bool = false,
|
||||
level3: bool = false,
|
||||
control: bool = false,
|
||||
};
|
||||
|
||||
/// The result of a lookup: the X11 `keysym`, and the `character` it produces (a Unicode
|
||||
/// scalar) when it is a printable key — null for keys that produce none (Return, F1, a
|
||||
/// bare dead key, an unmapped key).
|
||||
pub const Mapping = struct {
|
||||
keysym: u32,
|
||||
character: ?u21,
|
||||
};
|
||||
|
||||
/// Which level (0..3) a key of `kind` selects under `mods`. XKB's canonical semantics:
|
||||
/// Shift picks the odd level, AltGr (level3) adds 2, and Caps acts like Shift for the
|
||||
/// alphabetic types. See the XKB "key types" — this covers the ones the vendored layouts
|
||||
/// use; anything else falls back to shift-or-not.
|
||||
fn selectLevel(kind: KeyType, mods: Modifiers) usize {
|
||||
const shift_or_caps = mods.shift != mods.caps_lock; // XOR: Caps behaves like Shift
|
||||
const low: usize = if (mods.shift) 1 else 0;
|
||||
const high: usize = if (mods.level3) 2 else 0;
|
||||
return switch (kind) {
|
||||
.one_level => 0,
|
||||
.two_level, .keypad, .other => low,
|
||||
.alphabetic => if (shift_or_caps) 1 else 0,
|
||||
.four_level => low + high,
|
||||
.four_level_alphabetic => (if (shift_or_caps) @as(usize, 1) else 0) + high,
|
||||
// Caps affects only the base pair, not the AltGr pair.
|
||||
.four_level_semialphabetic => if (mods.level3) 2 + low else (if (shift_or_caps) @as(usize, 1) else 0),
|
||||
};
|
||||
}
|
||||
|
||||
/// Map a physical key (`hid_usage`, a USB HID keyboard-page usage) under `mods` on
|
||||
/// `layout` to its keysym and character. Falls back gracefully when the selected level is
|
||||
/// undefined for the key: it drops the AltGr component, then the shift component, so a key
|
||||
/// with only a base/shift pair still yields something sensible under AltGr.
|
||||
pub fn map(layout: *const Layout, hid_usage: u8, mods: Modifiers) Mapping {
|
||||
const key = &layout.keys[hid_usage];
|
||||
var level = selectLevel(key.kind, mods);
|
||||
// Fall back to a defined level: full -> without AltGr -> base.
|
||||
if (key.levels[level].keysym == 0 and key.levels[level].unicode == 0) {
|
||||
const candidates = [_]usize{ level & 1, 0 };
|
||||
for (candidates) |candidate| {
|
||||
if (key.levels[candidate].keysym != 0 or key.levels[candidate].unicode != 0) {
|
||||
level = candidate;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const chosen = key.levels[level];
|
||||
return .{
|
||||
.keysym = chosen.keysym,
|
||||
.character = if (chosen.unicode != 0) @intCast(chosen.unicode) else null,
|
||||
};
|
||||
}
|
||||
|
||||
/// Look up a layout by its name (`"us"`, `"gb"`, ...), or null if unknown.
|
||||
pub fn byName(name: []const u8) ?*const Layout {
|
||||
for (all) |layout| {
|
||||
if (std.mem.eql(u8, layout.name, name)) return layout;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests (host-run via `zig build test`) ---------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// USB HID usages used in the tests (keyboard page 0x07).
|
||||
const hid_a: u8 = 0x04;
|
||||
const hid_1: u8 = 0x1e;
|
||||
const hid_3: u8 = 0x20;
|
||||
|
||||
test "us: letters obey shift and caps" {
|
||||
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .shift = true }).character);
|
||||
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .caps_lock = true }).character);
|
||||
// Shift + Caps cancels for an alphabetic key.
|
||||
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{ .shift = true, .caps_lock = true }).character);
|
||||
}
|
||||
|
||||
test "us: digits and their shifted symbols" {
|
||||
try testing.expectEqual(@as(?u21, '1'), map(us, hid_1, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, '!'), map(us, hid_1, .{ .shift = true }).character);
|
||||
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||
// A digit is not alphabetic: Caps alone must not shift it.
|
||||
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{ .caps_lock = true }).character);
|
||||
}
|
||||
|
||||
test "layouts differ: GB pound vs US hash on shift+3" {
|
||||
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||
try testing.expectEqual(@as(?u21, '£'), map(gb, hid_3, .{ .shift = true }).character);
|
||||
}
|
||||
|
||||
test "french azerty places q where us has a" {
|
||||
try testing.expectEqual(@as(?u21, 'q'), map(fr, hid_a, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, 'Q'), map(fr, hid_a, .{ .shift = true }).character);
|
||||
}
|
||||
|
||||
test "byName resolves and rejects" {
|
||||
try testing.expect(byName("us") == us);
|
||||
try testing.expect(byName("gb") == gb);
|
||||
try testing.expect(byName("nonsense") == null);
|
||||
}
|
||||
|
||||
test "unmapped key yields no character" {
|
||||
// HID 0x00 is not a key; every level is empty.
|
||||
try testing.expectEqual(@as(?u21, null), map(us, 0x00, .{}).character);
|
||||
}
|
||||
-1127
File diff suppressed because it is too large
Load Diff
@@ -1,208 +0,0 @@
|
||||
//! AML (ACPI Machine Language) — the bytecode in the DSDT and SSDTs that describes
|
||||
//! the parts of the machine the static tables don't.
|
||||
//!
|
||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||
//! access — so callers can resolve device status (`_STA`), current resource
|
||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||
|
||||
const std = @import("std");
|
||||
const op = @import("opcodes.zig");
|
||||
const parser = @import("parser.zig");
|
||||
const namespace = @import("namespace.zig");
|
||||
const interp = @import("interp.zig");
|
||||
|
||||
pub const Namespace = namespace.Namespace;
|
||||
pub const Node = namespace.Node;
|
||||
pub const NodeKind = namespace.NodeKind;
|
||||
|
||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||
/// enough for device discovery. See `interp.zig`.
|
||||
pub const Interp = interp.Interp;
|
||||
pub const Object = interp.Object;
|
||||
pub const EvalHal = interp.Hal;
|
||||
|
||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||
pub const SleepType = struct {
|
||||
slp_typ_a: u8,
|
||||
slp_typ_b: u8,
|
||||
};
|
||||
|
||||
pub const ParseResult = struct {
|
||||
namespace: Namespace,
|
||||
/// Bytes the parser consumed across all blocks...
|
||||
consumed: usize,
|
||||
/// ...out of this many. A clean full traversal has `consumed == total`.
|
||||
total: usize,
|
||||
};
|
||||
|
||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||
var ns = try Namespace.init(allocator);
|
||||
var consumed: usize = 0;
|
||||
var total: usize = 0;
|
||||
for (blocks) |block| {
|
||||
var p = parser.Parser.init(block, &ns);
|
||||
consumed += p.parseAll();
|
||||
total += block.len;
|
||||
}
|
||||
return .{ .namespace = ns, .consumed = consumed, .total = total };
|
||||
}
|
||||
|
||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||
pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
||||
const seg = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||
const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null;
|
||||
if (node.kind != .name) return null;
|
||||
return parseSleepPackage(node.value);
|
||||
}
|
||||
|
||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||
if (value.len == 0 or value[0] != op.package_op) return null;
|
||||
var p: usize = 1;
|
||||
p += pkgLengthSize(value, p) orelse return null;
|
||||
if (p >= value.len) return null;
|
||||
const num_elements = value[p];
|
||||
p += 1;
|
||||
|
||||
const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||
}
|
||||
|
||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||
fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||
if (p >= bytes.len) return null;
|
||||
const follow: usize = bytes[p] >> 6;
|
||||
if (p + 1 + follow > bytes.len) return null;
|
||||
return 1 + follow;
|
||||
}
|
||||
|
||||
/// Read one AML integer data object at `p`, advancing `p`.
|
||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const opcode = bytes[p.*];
|
||||
p.* += 1;
|
||||
return switch (opcode) {
|
||||
op.zero_op => 0,
|
||||
op.one_op => 1,
|
||||
op.ones_op => 0xFF,
|
||||
op.byte_prefix => readLittle(bytes, p, 1),
|
||||
op.word_prefix => readLittle(bytes, p, 2),
|
||||
op.dword_prefix => readLittle(bytes, p, 4),
|
||||
op.qword_prefix => readLittle(bytes, p, 8),
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
fn readLittle(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
||||
if (p.* + n > bytes.len) return null;
|
||||
var v: u64 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < n) : (k += 1) v |= @as(u64, bytes[p.* + k]) << @intCast(k * 8);
|
||||
p.* += n;
|
||||
return v;
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
test "parses a nested namespace and finds the sleep package" {
|
||||
// A hand-assembled AML blob (all PkgLengths computed to be single-byte):
|
||||
// Name(_S5, Package(2){0x05, 0x00})
|
||||
// Scope(\_SB) { Device(PCI0) {
|
||||
// Name(_HID, 0x11)
|
||||
// Method(MTHD, 1) {}
|
||||
// Method(CALL, 0) { MTHD(Zero) } // invocation of a 1-arg method
|
||||
// } }
|
||||
// OperationRegion(DBG0, SystemIO, 0x0402, 1)
|
||||
// Field(DBG0, ...) { DBGB, 8 }
|
||||
const blob = [_]u8{
|
||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||
// Scope(\_SB) pkglen=0x27
|
||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||
// Device(PCI0) pkglen=0x1F
|
||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
||||
// Name(_HID, 0x11)
|
||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||
// Method(MTHD, flags=1) empty, pkglen=0x06
|
||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
||||
// Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B
|
||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
||||
// Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B
|
||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
|
||||
// Integrity: the parser consumed exactly the whole blob (no desync).
|
||||
try std.testing.expectEqual(blob.len, result.consumed);
|
||||
try std.testing.expectEqual(blob.len, result.total);
|
||||
|
||||
const ns = &result.namespace;
|
||||
|
||||
// Expected top-level nodes.
|
||||
const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||
const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||
_ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||
|
||||
// The 1-arg method's arg count was parsed from its flags byte.
|
||||
const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||
|
||||
// OperationRegion and the Field unit made it into the namespace.
|
||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||
|
||||
// The sleep package decoded.
|
||||
const s5 = sleepState(ns, 5) orelse return error.NoS5;
|
||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||
}
|
||||
|
||||
fn noMap(_: u64, _: u64, _: bool) void {}
|
||||
fn noRead(_: u8, _: u16) u32 {
|
||||
return 0;
|
||||
}
|
||||
fn noWrite(_: u8, _: u16, _: u32) void {}
|
||||
|
||||
test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||
// Method(TST_, 1) {
|
||||
// Store(Arg0, Local0); Add(Local0, 5, Local0)
|
||||
// If (LGreater(Local0, 10)) { Return(One) }
|
||||
// Return(Zero)
|
||||
// }
|
||||
const blob = [_]u8{
|
||||
0x14, 0x18, 0x54, 0x53, 0x54, 0x5F, 0x01, // Method TST_, 1 arg
|
||||
0x70, 0x68, 0x60, // Store(Arg0, Local0)
|
||||
0x72, 0x60, 0x0A, 0x05, 0x60, // Add(Local0, 5, Local0)
|
||||
0xA0, 0x07, 0x94, 0x60, 0x0A, 0x0A, 0xA4, 0x01, // If(LGreater(Local0,10)) { Return(One) }
|
||||
0xA4, 0x00, // Return(Zero)
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
const ns = &result.namespace;
|
||||
const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||
|
||||
var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||
|
||||
const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInt());
|
||||
const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInt());
|
||||
}
|
||||
@@ -1,737 +0,0 @@
|
||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||
//! structural namespace. It executes control methods (their bodies captured by
|
||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||
//! flow, locals/args, and
|
||||
//! OperationRegion field access those methods reach for.
|
||||
//!
|
||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||
//! access; CreateField buffer patching (the common current-resource-settings
|
||||
//! (`_CRS`) idiom); and field
|
||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||
//! fall back — never a hard failure.
|
||||
|
||||
const std = @import("std");
|
||||
const op = @import("opcodes.zig");
|
||||
const nsp = @import("namespace.zig");
|
||||
const Node = nsp.Node;
|
||||
const Namespace = nsp.Namespace;
|
||||
|
||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
||||
pub const Hal = struct {
|
||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
||||
pioRead: *const fn (width: u8, port: u16) u32,
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||
|
||||
/// A runtime AML value.
|
||||
pub const Object = union(enum) {
|
||||
uninitialized,
|
||||
integer: u64,
|
||||
buffer: []u8,
|
||||
string: []u8,
|
||||
package: []Object,
|
||||
reference: *Node,
|
||||
|
||||
pub fn asInt(self: Object) Error!u64 {
|
||||
return switch (self) {
|
||||
.integer => |v| v,
|
||||
.buffer => |b| blk: {
|
||||
var v: u64 = 0;
|
||||
for (b, 0..) |byte, i| {
|
||||
if (i >= 8) break;
|
||||
v |= @as(u64, byte) << @intCast(i * 8);
|
||||
}
|
||||
break :blk v;
|
||||
},
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
const max_segs = 16;
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segs: [max_segs][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segs[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
const Cursor = struct {
|
||||
b: []const u8,
|
||||
i: usize = 0,
|
||||
|
||||
fn eof(self: *Cursor) bool {
|
||||
return self.i >= self.b.len;
|
||||
}
|
||||
fn peek(self: *Cursor) ?u8 {
|
||||
return if (self.eof()) null else self.b[self.i];
|
||||
}
|
||||
fn byte(self: *Cursor) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const v = self.b[self.i];
|
||||
self.i += 1;
|
||||
return v;
|
||||
}
|
||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||
if (self.i + n > self.b.len) return error.Truncated;
|
||||
const s = self.b[self.i .. self.i + n];
|
||||
self.i += n;
|
||||
return s;
|
||||
}
|
||||
fn pkgLen(self: *Cursor) Error!usize {
|
||||
const lead = try self.byte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var k: usize = 0;
|
||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||
return value;
|
||||
}
|
||||
fn nameString(self: *Cursor) Error!NamePath {
|
||||
var np = NamePath{};
|
||||
if (self.peek() == op.root_char) {
|
||||
np.rooted = true;
|
||||
self.i += 1;
|
||||
} else {
|
||||
while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1;
|
||||
}
|
||||
const lead = self.peek() orelse return np;
|
||||
switch (lead) {
|
||||
0x00 => self.i += 1,
|
||||
op.dual_name_prefix => {
|
||||
self.i += 1;
|
||||
try self.seg(&np);
|
||||
try self.seg(&np);
|
||||
},
|
||||
op.multi_name_prefix => {
|
||||
self.i += 1;
|
||||
const cnt = try self.byte();
|
||||
var k: usize = 0;
|
||||
while (k < cnt) : (k += 1) try self.seg(&np);
|
||||
},
|
||||
else => try self.seg(&np),
|
||||
}
|
||||
return np;
|
||||
}
|
||||
fn seg(self: *Cursor, np: *NamePath) Error!void {
|
||||
const s = try self.take(4);
|
||||
if (np.count < max_segs) {
|
||||
np.segs[np.count] = s[0..4].*;
|
||||
np.count += 1;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const Frame = struct {
|
||||
args: [7]Object = .{.uninitialized} ** 7,
|
||||
locals: [8]Object = .{.uninitialized} ** 8,
|
||||
scope: *Node,
|
||||
ret: Object = .uninitialized,
|
||||
returned: bool = false,
|
||||
broke: bool = false,
|
||||
};
|
||||
|
||||
/// A CreateField binding: a name that indexes into a buffer object.
|
||||
const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 };
|
||||
|
||||
pub const Interp = struct {
|
||||
ns: *Namespace,
|
||||
hal: Hal,
|
||||
arena: std.mem.Allocator,
|
||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||
dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||
/// CreateField bindings active for the current evaluation.
|
||||
fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{},
|
||||
|
||||
pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp {
|
||||
return .{ .ns = ns, .hal = hal, .arena = arena };
|
||||
}
|
||||
|
||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||
/// Field. Resets per-evaluation runtime state first.
|
||||
pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
||||
self.dyn.clearRetainingCapacity();
|
||||
self.fields.clearRetainingCapacity();
|
||||
return self.invoke(node, args);
|
||||
}
|
||||
|
||||
fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var frame = Frame{ .scope = node };
|
||||
for (args, 0..) |a, i| {
|
||||
if (i < frame.args.len) frame.args[i] = a;
|
||||
}
|
||||
var cur = Cursor{ .b = node.value };
|
||||
try self.execList(&cur, &frame);
|
||||
return frame.ret;
|
||||
},
|
||||
.name => {
|
||||
if (self.dyn.get(node)) |o| return o;
|
||||
var cur = Cursor{ .b = node.value };
|
||||
var frame = Frame{ .scope = node.parent orelse self.ns.root };
|
||||
return self.term(&cur, &frame);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||
fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void {
|
||||
while (!cur.eof() and !frame.returned and !frame.broke) {
|
||||
_ = try self.term(cur, frame);
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||
/// statements).
|
||||
fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const lead = cur.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameRef(cur, frame);
|
||||
_ = try cur.byte();
|
||||
|
||||
return switch (lead) {
|
||||
op.zero_op => Object{ .integer = 0 },
|
||||
op.one_op => Object{ .integer = 1 },
|
||||
op.ones_op => Object{ .integer = ~@as(u64, 0) },
|
||||
op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) },
|
||||
op.word_prefix => Object{ .integer = try self.readConst(cur, 2) },
|
||||
op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) },
|
||||
op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) },
|
||||
op.string_prefix => try self.readString(cur),
|
||||
op.buffer_op => try self.buffer(cur, frame),
|
||||
op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op),
|
||||
|
||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op],
|
||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op],
|
||||
|
||||
op.return_op => blk: {
|
||||
frame.ret = try self.term(cur, frame);
|
||||
frame.returned = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
op.break_op => blk: {
|
||||
frame.broke = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
op.continue_op, op.noop_op => .uninitialized,
|
||||
|
||||
op.if_op => try self.ifElse(cur, frame),
|
||||
op.while_op => try self.whileLoop(cur, frame),
|
||||
op.store_op => try self.store(cur, frame),
|
||||
op.increment_op => try self.incDec(cur, frame, 1),
|
||||
op.decrement_op => try self.incDec(cur, frame, -1),
|
||||
|
||||
op.add_op => try self.binary(cur, frame, .add),
|
||||
op.subtract_op => try self.binary(cur, frame, .sub),
|
||||
op.multiply_op => try self.binary(cur, frame, .mul),
|
||||
op.mod_op => try self.binary(cur, frame, .mod),
|
||||
op.and_op => try self.binary(cur, frame, .band),
|
||||
op.or_op => try self.binary(cur, frame, .bor),
|
||||
op.xor_op => try self.binary(cur, frame, .bxor),
|
||||
op.nand_op => try self.binary(cur, frame, .nand),
|
||||
op.nor_op => try self.binary(cur, frame, .nor),
|
||||
op.shift_left_op => try self.binary(cur, frame, .shl),
|
||||
op.shift_right_op => try self.binary(cur, frame, .shr),
|
||||
op.divide_op => try self.divide(cur, frame),
|
||||
|
||||
op.land_op => try self.logic2(cur, frame, .land),
|
||||
op.lor_op => try self.logic2(cur, frame, .lor),
|
||||
op.lequal_op => try self.logic2(cur, frame, .eq),
|
||||
op.lgreater_op => try self.logic2(cur, frame, .gt),
|
||||
op.lless_op => try self.logic2(cur, frame, .lt),
|
||||
op.lnot_op => try self.lnot(cur, frame),
|
||||
|
||||
op.not_op => blk: {
|
||||
const v = try self.evalInt(cur, frame);
|
||||
const r = ~v;
|
||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
||||
break :blk .{ .integer = r };
|
||||
},
|
||||
|
||||
op.size_of_op => try self.sizeOf(cur, frame),
|
||||
op.index_op => try self.index(cur, frame),
|
||||
op.deref_of_op => try self.derefOf(cur, frame),
|
||||
op.to_integer_op => blk: {
|
||||
const v = try self.evalInt(cur, frame);
|
||||
try self.storeTarget(cur, frame, .{ .integer = v });
|
||||
break :blk .{ .integer = v };
|
||||
},
|
||||
op.to_buffer_op => try self.passThroughUnary(cur, frame),
|
||||
|
||||
op.ext_op_prefix => try self.ext(cur, frame),
|
||||
|
||||
// CreateXField: source, index, name (bit widths differ by op)
|
||||
op.create_bit_field_op => try self.createField(cur, frame, 1),
|
||||
op.create_byte_field_op => try self.createField(cur, frame, 8),
|
||||
op.create_word_field_op => try self.createField(cur, frame, 16),
|
||||
op.create_dword_field_op => try self.createField(cur, frame, 32),
|
||||
op.create_qword_field_op => try self.createField(cur, frame, 64),
|
||||
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
|
||||
// --- name references ----------------------------------------------------
|
||||
|
||||
fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const np = try cur.nameString();
|
||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse
|
||||
return .uninitialized; // unknown name -> treat as uninitialised
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var argbuf: [7]Object = undefined;
|
||||
var i: usize = 0;
|
||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame);
|
||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
.name => return self.invoke(node, &.{}),
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
// --- data objects -------------------------------------------------------
|
||||
|
||||
fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 {
|
||||
_ = self;
|
||||
const bytes = try cur.take(n);
|
||||
var v: u64 = 0;
|
||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn readString(self: *Interp, cur: *Cursor) Error!Object {
|
||||
const start = cur.i;
|
||||
while (cur.peek()) |c| {
|
||||
cur.i += 1;
|
||||
if (c == 0) break;
|
||||
}
|
||||
const raw = cur.b[start .. cur.i - 1];
|
||||
const s = try self.arena.dupe(u8, raw);
|
||||
return .{ .string = s };
|
||||
}
|
||||
|
||||
fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = cur.i;
|
||||
const len = try cur.pkgLen();
|
||||
const end = @min(start + len, cur.b.len);
|
||||
const size = try self.evalInt(cur, frame);
|
||||
const data = cur.b[@min(cur.i, end)..end];
|
||||
const buf = try self.arena.alloc(u8, @intCast(size));
|
||||
@memset(buf, 0);
|
||||
@memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]);
|
||||
cur.i = end;
|
||||
return .{ .buffer = buf };
|
||||
}
|
||||
|
||||
fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||
const start = cur.i;
|
||||
const len = try cur.pkgLen();
|
||||
const end = @min(start + len, cur.b.len);
|
||||
const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte();
|
||||
const elems = try self.arena.alloc(Object, count);
|
||||
var i: usize = 0;
|
||||
while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame);
|
||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||
cur.i = end;
|
||||
return .{ .package = elems };
|
||||
}
|
||||
|
||||
// --- operators ----------------------------------------------------------
|
||||
|
||||
const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||
|
||||
fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object {
|
||||
const a = try self.evalInt(cur, frame);
|
||||
const b = try self.evalInt(cur, frame);
|
||||
const r: u64 = switch (kind) {
|
||||
.add => a +% b,
|
||||
.sub => a -% b,
|
||||
.mul => a *% b,
|
||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||
.band => a & b,
|
||||
.bor => a | b,
|
||||
.bxor => a ^ b,
|
||||
.nand => ~(a & b),
|
||||
.nor => ~(a | b),
|
||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||
};
|
||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const a = try self.evalInt(cur, frame);
|
||||
const b = try self.evalInt(cur, frame);
|
||||
if (b == 0) return error.DivByZero;
|
||||
try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target
|
||||
try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target
|
||||
return .{ .integer = a / b };
|
||||
}
|
||||
|
||||
const LogicOp = enum { land, lor, eq, gt, lt };
|
||||
|
||||
fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object {
|
||||
const a = try self.evalInt(cur, frame);
|
||||
const b = try self.evalInt(cur, frame);
|
||||
const r = switch (kind) {
|
||||
.land => a != 0 and b != 0,
|
||||
.lor => a != 0 or b != 0,
|
||||
.eq => a == b,
|
||||
.gt => a > b,
|
||||
.lt => a < b,
|
||||
};
|
||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||
}
|
||||
|
||||
fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
// 0x92 0x93/94/95 are the compound comparisons.
|
||||
const b = cur.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
op.lnot.not_equal => {
|
||||
cur.i += 1;
|
||||
const x = try self.evalInt(cur, frame);
|
||||
const y = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
op.lnot.less_equal => {
|
||||
cur.i += 1;
|
||||
const x = try self.evalInt(cur, frame);
|
||||
const y = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
op.lnot.greater_equal => {
|
||||
cur.i += 1;
|
||||
const x = try self.evalInt(cur, frame);
|
||||
const y = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
else => {
|
||||
const x = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||
// Operand is a SuperName that is both read and written.
|
||||
const save = cur.i;
|
||||
const cur_val = try self.term(cur, frame);
|
||||
const v = try cur_val.asInt();
|
||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||
var tcur = Cursor{ .b = cur.b, .i = save };
|
||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(cur, frame);
|
||||
return .{ .integer = switch (o) {
|
||||
.buffer => |b| b.len,
|
||||
.string => |s| s.len,
|
||||
.package => |p| p.len,
|
||||
else => 0,
|
||||
} };
|
||||
}
|
||||
|
||||
fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(cur, frame);
|
||||
try self.storeTarget(cur, frame, o);
|
||||
return o;
|
||||
}
|
||||
|
||||
fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const src = try self.term(cur, frame);
|
||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
||||
// Optional target (a reference); we don't materialise references, so store
|
||||
// the indexed value if a target is present.
|
||||
const val: Object = switch (src) {
|
||||
.buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 },
|
||||
.package => |p| if (idx < p.len) p[idx] else .uninitialized,
|
||||
.string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 },
|
||||
else => .uninitialized,
|
||||
};
|
||||
try self.storeTarget(cur, frame, val);
|
||||
return val;
|
||||
}
|
||||
|
||||
fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(cur, frame);
|
||||
return switch (o) {
|
||||
.reference => |n| self.invoke(n, &.{}),
|
||||
else => o,
|
||||
};
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = cur.i;
|
||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
||||
const cond = try self.evalInt(cur, frame);
|
||||
if (cond != 0) {
|
||||
var body = Cursor{ .b = cur.b[0..end], .i = cur.i };
|
||||
try self.execList(&body, frame);
|
||||
cur.i = end;
|
||||
// Skip a trailing Else.
|
||||
if (cur.peek() == op.else_op) {
|
||||
cur.i += 1;
|
||||
const es = cur.i;
|
||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
||||
cur.i = ee;
|
||||
}
|
||||
} else {
|
||||
cur.i = end;
|
||||
if (cur.peek() == op.else_op) {
|
||||
cur.i += 1;
|
||||
const es = cur.i;
|
||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
||||
var body = Cursor{ .b = cur.b[0..ee], .i = cur.i };
|
||||
try self.execList(&body, frame);
|
||||
cur.i = ee;
|
||||
}
|
||||
}
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = cur.i;
|
||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
||||
const pred_at = cur.i;
|
||||
var guard: usize = 0;
|
||||
while (guard < 100_000) : (guard += 1) {
|
||||
var pc = Cursor{ .b = cur.b[0..end], .i = pred_at };
|
||||
const cond = try self.evalInt(&pc, frame);
|
||||
if (cond == 0) break;
|
||||
var body = Cursor{ .b = cur.b[0..end], .i = pc.i };
|
||||
try self.execList(&body, frame);
|
||||
if (frame.returned) break;
|
||||
if (frame.broke) {
|
||||
frame.broke = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
cur.i = end;
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
// --- store --------------------------------------------------------------
|
||||
|
||||
fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const value = try self.term(cur, frame);
|
||||
try self.storeInto(cur, frame, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
/// A Store *target* that may be NullName (no store).
|
||||
fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
if (cur.peek() == 0x00) {
|
||||
cur.i += 1; // NullName
|
||||
return;
|
||||
}
|
||||
try self.storeInto(cur, frame, value);
|
||||
}
|
||||
|
||||
fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
const lead = cur.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) {
|
||||
const np = try cur.nameString();
|
||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return;
|
||||
if (self.fields.get(node)) |bf| {
|
||||
try self.writeBufField(bf, try value.asInt());
|
||||
} else if (node.kind == .field) {
|
||||
try self.writeField(node, try value.asInt());
|
||||
} else {
|
||||
try self.dyn.put(self.arena, node, value);
|
||||
}
|
||||
return;
|
||||
}
|
||||
_ = try cur.byte();
|
||||
switch (lead) {
|
||||
0x00 => {}, // NullName
|
||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value,
|
||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value,
|
||||
op.index_op => {
|
||||
const src = try self.term(cur, frame);
|
||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
||||
switch (src) {
|
||||
.buffer => |b| if (idx < b.len) {
|
||||
b[idx] = @truncate(try value.asInt());
|
||||
},
|
||||
.package => |p| if (idx < p.len) {
|
||||
p[idx] = value;
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- CreateField (buffer patching) --------------------------------------
|
||||
|
||||
fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||
const src = try self.term(cur, frame); // source buffer (as a reference or value)
|
||||
const bit_index = try self.evalInt(cur, frame);
|
||||
const np = try cur.nameString();
|
||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized;
|
||||
|
||||
// Bind the new name to the source buffer's node so stores land in it.
|
||||
const buf_node: *Node = switch (src) {
|
||||
.reference => |n| n,
|
||||
else => return .uninitialized,
|
||||
};
|
||||
// Materialise the buffer into `dyn` so patches persist and are returned.
|
||||
if (self.dyn.get(buf_node) == null) {
|
||||
const val = try self.invoke(buf_node, &.{});
|
||||
try self.dyn.put(self.arena, buf_node, val);
|
||||
}
|
||||
const byte_off: usize = @intCast(bit_index / 8);
|
||||
try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void {
|
||||
const obj = self.dyn.get(bf.buf) orelse return;
|
||||
const buf = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
const nbytes = (bf.bit_width + 7) / 8;
|
||||
var k: usize = 0;
|
||||
while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) {
|
||||
buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||
}
|
||||
}
|
||||
|
||||
// --- OperationRegion field access ---------------------------------------
|
||||
|
||||
fn readField(self: *Interp, field: *Node) Error!u64 {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const nbytes = (total + 7) / 8;
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < nbytes) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||
return @truncate(masked);
|
||||
}
|
||||
|
||||
fn writeField(self: *Interp, field: *Node, value: u64) Error!void {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const nbytes = (total + 7) / 8;
|
||||
// Read-modify-write byte by byte.
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < nbytes) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const mask = bitMask(field.bit_width) << shift;
|
||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||
k = 0;
|
||||
while (k < nbytes) : (k += 1) {
|
||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||
}
|
||||
}
|
||||
|
||||
fn regionBase(self: *Interp, region: *Node) Error!u64 {
|
||||
var cur = Cursor{ .b = region.region_offset_aml };
|
||||
var frame = Frame{ .scope = region.parent orelse self.ns.root };
|
||||
return (try self.term(&cur, &frame)).asInt();
|
||||
}
|
||||
|
||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
||||
switch (space) {
|
||||
0 => { // SystemMemory
|
||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
||||
const p: *align(1) const volatile u8 = @ptrFromInt(addr);
|
||||
return p.*;
|
||||
},
|
||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
||||
switch (space) {
|
||||
0 => {
|
||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
||||
const p: *align(1) volatile u8 = @ptrFromInt(addr);
|
||||
p.* = value;
|
||||
},
|
||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- extended opcodes ---------------------------------------------------
|
||||
|
||||
fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const e = try cur.byte();
|
||||
switch (e) {
|
||||
op.ext.debug => return .uninitialized,
|
||||
op.ext.revision => return .{ .integer = 2 },
|
||||
op.ext.timer => return .{ .integer = 0 },
|
||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||
op.ext.acquire => {
|
||||
_ = try self.term(cur, frame); // mutex SuperName
|
||||
_ = try cur.take(2); // timeout
|
||||
return .{ .integer = 0 }; // acquired
|
||||
},
|
||||
op.ext.release, op.ext.reset, op.ext.signal => {
|
||||
_ = try self.term(cur, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
op.ext.wait => {
|
||||
_ = try self.term(cur, frame);
|
||||
_ = try self.term(cur, frame);
|
||||
return .{ .integer = 0 };
|
||||
},
|
||||
op.ext.sleep, op.ext.stall => {
|
||||
_ = try self.term(cur, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 {
|
||||
return (try self.term(cur, frame)).asInt();
|
||||
}
|
||||
};
|
||||
|
||||
fn bitMask(width: u32) u128 {
|
||||
if (width >= 128) return ~@as(u128, 0);
|
||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||
}
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
||||
b == op.name_char_underscore or
|
||||
b == op.root_char or
|
||||
b == op.parent_prefix_char or
|
||||
b == op.dual_name_prefix or
|
||||
b == op.multi_name_prefix;
|
||||
}
|
||||
@@ -1,137 +0,0 @@
|
||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||
//!
|
||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||
//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`).
|
||||
|
||||
// --- name / path characters -------------------------------------------------
|
||||
pub const zero_op = 0x00;
|
||||
pub const one_op = 0x01;
|
||||
pub const alias_op = 0x06;
|
||||
pub const name_op = 0x08;
|
||||
pub const byte_prefix = 0x0A;
|
||||
pub const word_prefix = 0x0B;
|
||||
pub const dword_prefix = 0x0C;
|
||||
pub const string_prefix = 0x0D;
|
||||
pub const qword_prefix = 0x0E;
|
||||
pub const scope_op = 0x10;
|
||||
pub const buffer_op = 0x11;
|
||||
pub const package_op = 0x12;
|
||||
pub const var_package_op = 0x13;
|
||||
pub const method_op = 0x14;
|
||||
pub const external_op = 0x15;
|
||||
|
||||
pub const dual_name_prefix = 0x2E;
|
||||
pub const multi_name_prefix = 0x2F;
|
||||
pub const ext_op_prefix = 0x5B;
|
||||
pub const root_char = 0x5C;
|
||||
pub const parent_prefix_char = 0x5E;
|
||||
pub const name_char_underscore = 0x5F;
|
||||
|
||||
pub const digit_char_start = 0x30;
|
||||
pub const digit_char_end = 0x39;
|
||||
pub const name_char_start = 0x41; // 'A'
|
||||
pub const name_char_end = 0x5A; // 'Z'
|
||||
|
||||
// --- locals / args ----------------------------------------------------------
|
||||
pub const local0_op = 0x60;
|
||||
pub const local7_op = 0x67;
|
||||
pub const arg0_op = 0x68;
|
||||
pub const arg6_op = 0x6E;
|
||||
|
||||
// --- store / references / arithmetic ---------------------------------------
|
||||
pub const store_op = 0x70;
|
||||
pub const ref_of_op = 0x71;
|
||||
pub const add_op = 0x72;
|
||||
pub const concat_op = 0x73;
|
||||
pub const subtract_op = 0x74;
|
||||
pub const increment_op = 0x75;
|
||||
pub const decrement_op = 0x76;
|
||||
pub const multiply_op = 0x77;
|
||||
pub const divide_op = 0x78;
|
||||
pub const shift_left_op = 0x79;
|
||||
pub const shift_right_op = 0x7A;
|
||||
pub const and_op = 0x7B;
|
||||
pub const nand_op = 0x7C;
|
||||
pub const or_op = 0x7D;
|
||||
pub const nor_op = 0x7E;
|
||||
pub const xor_op = 0x7F;
|
||||
pub const not_op = 0x80;
|
||||
pub const find_set_left_bit_op = 0x81;
|
||||
pub const find_set_right_bit_op = 0x82;
|
||||
pub const deref_of_op = 0x83;
|
||||
pub const concat_res_op = 0x84;
|
||||
pub const mod_op = 0x85;
|
||||
pub const notify_op = 0x86;
|
||||
pub const size_of_op = 0x87;
|
||||
pub const index_op = 0x88;
|
||||
pub const match_op = 0x89;
|
||||
pub const create_dword_field_op = 0x8A;
|
||||
pub const create_word_field_op = 0x8B;
|
||||
pub const create_byte_field_op = 0x8C;
|
||||
pub const create_bit_field_op = 0x8D;
|
||||
pub const object_type_op = 0x8E;
|
||||
pub const create_qword_field_op = 0x8F;
|
||||
|
||||
pub const land_op = 0x90;
|
||||
pub const lor_op = 0x91;
|
||||
pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`)
|
||||
pub const lequal_op = 0x93;
|
||||
pub const lgreater_op = 0x94;
|
||||
pub const lless_op = 0x95;
|
||||
pub const to_buffer_op = 0x96;
|
||||
pub const to_decimal_string_op = 0x97;
|
||||
pub const to_hex_string_op = 0x98;
|
||||
pub const to_integer_op = 0x99;
|
||||
pub const to_string_op = 0x9C;
|
||||
pub const copy_object_op = 0x9D;
|
||||
pub const mid_op = 0x9E;
|
||||
pub const continue_op = 0x9F;
|
||||
pub const if_op = 0xA0;
|
||||
pub const else_op = 0xA1;
|
||||
pub const while_op = 0xA2;
|
||||
pub const noop_op = 0xA3;
|
||||
pub const return_op = 0xA4;
|
||||
pub const break_op = 0xA5;
|
||||
pub const break_point_op = 0xCC;
|
||||
pub const ones_op = 0xFF;
|
||||
|
||||
/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes.
|
||||
pub const lnot = struct {
|
||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||
};
|
||||
|
||||
/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B).
|
||||
pub const ext = struct {
|
||||
pub const mutex = 0x01;
|
||||
pub const event = 0x02;
|
||||
pub const cond_ref_of = 0x12;
|
||||
pub const create_field = 0x13;
|
||||
pub const load_table = 0x1F;
|
||||
pub const load = 0x20;
|
||||
pub const stall = 0x21;
|
||||
pub const sleep = 0x22;
|
||||
pub const acquire = 0x23;
|
||||
pub const signal = 0x24;
|
||||
pub const wait = 0x25;
|
||||
pub const reset = 0x26;
|
||||
pub const release = 0x27;
|
||||
pub const from_bcd = 0x28;
|
||||
pub const to_bcd = 0x29;
|
||||
pub const unload = 0x2A;
|
||||
pub const revision = 0x30;
|
||||
pub const debug = 0x31;
|
||||
pub const fatal = 0x32;
|
||||
pub const timer = 0x33;
|
||||
pub const op_region = 0x80;
|
||||
pub const field = 0x81;
|
||||
pub const device = 0x82;
|
||||
pub const processor = 0x83;
|
||||
pub const power_res = 0x84;
|
||||
pub const thermal_zone = 0x85;
|
||||
pub const index_field = 0x86;
|
||||
pub const bank_field = 0x87;
|
||||
pub const data_region = 0x88;
|
||||
};
|
||||
@@ -1,517 +0,0 @@
|
||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||
//! cursor stays aligned and every named object is recorded.
|
||||
//!
|
||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||
//! operand position is a call whose argument count is only known from the method's
|
||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||
//!
|
||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||
|
||||
const std = @import("std");
|
||||
const op = @import("opcodes.zig");
|
||||
const ns = @import("namespace.zig");
|
||||
const Namespace = ns.Namespace;
|
||||
const Node = ns.Node;
|
||||
|
||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||
|
||||
const max_segs = 64;
|
||||
|
||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||
/// of 4-byte segments.
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segs: [max_segs][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segs[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
pub const Parser = struct {
|
||||
aml: []const u8,
|
||||
pos: usize = 0,
|
||||
namespace: *Namespace,
|
||||
|
||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||
return .{ .aml = aml, .namespace = namespace };
|
||||
}
|
||||
|
||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||
pub fn parseAll(self: *Parser) usize {
|
||||
self.termList(self.aml.len, self.namespace.root);
|
||||
return self.pos;
|
||||
}
|
||||
|
||||
// --- cursor primitives --------------------------------------------------
|
||||
|
||||
fn eof(self: *Parser) bool {
|
||||
return self.pos >= self.aml.len;
|
||||
}
|
||||
|
||||
fn peek(self: *Parser) ?u8 {
|
||||
return if (self.eof()) null else self.aml[self.pos];
|
||||
}
|
||||
|
||||
fn readByte(self: *Parser) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const b = self.aml[self.pos];
|
||||
self.pos += 1;
|
||||
return b;
|
||||
}
|
||||
|
||||
fn skip(self: *Parser, n: usize) Error!void {
|
||||
if (self.pos + n > self.aml.len) return error.Truncated;
|
||||
self.pos += n;
|
||||
}
|
||||
|
||||
fn skipCString(self: *Parser) Error!void {
|
||||
while (true) {
|
||||
const b = try self.readByte();
|
||||
if (b == 0) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||
/// follow; the value counts from the start of the PkgLength field.
|
||||
fn readPkgLength(self: *Parser) Error!usize {
|
||||
const lead = try self.readByte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var i: usize = 0;
|
||||
while (i < follow) : (i += 1) {
|
||||
const b = try self.readByte();
|
||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
fn readNameSeg(self: *Parser) Error![4]u8 {
|
||||
if (self.pos + 4 > self.aml.len) return error.Truncated;
|
||||
const seg = self.aml[self.pos..][0..4].*;
|
||||
self.pos += 4;
|
||||
return seg;
|
||||
}
|
||||
|
||||
fn readNameString(self: *Parser) Error!NamePath {
|
||||
var np = NamePath{};
|
||||
// A NameString is either root-anchored or parent-relative, not both.
|
||||
if (self.peek() == op.root_char) {
|
||||
np.rooted = true;
|
||||
self.pos += 1;
|
||||
} else {
|
||||
while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1;
|
||||
}
|
||||
|
||||
const lead = self.peek() orelse return np;
|
||||
switch (lead) {
|
||||
0x00 => self.pos += 1, // NullName
|
||||
op.dual_name_prefix => {
|
||||
self.pos += 1;
|
||||
try self.appendSeg(&np);
|
||||
try self.appendSeg(&np);
|
||||
},
|
||||
op.multi_name_prefix => {
|
||||
self.pos += 1;
|
||||
const cnt = try self.readByte();
|
||||
var i: usize = 0;
|
||||
while (i < cnt) : (i += 1) try self.appendSeg(&np);
|
||||
},
|
||||
else => {
|
||||
if (isNameStart(lead)) try self.appendSeg(&np);
|
||||
},
|
||||
}
|
||||
return np;
|
||||
}
|
||||
|
||||
fn appendSeg(self: *Parser, np: *NamePath) Error!void {
|
||||
const seg = try self.readNameSeg();
|
||||
if (np.count < max_segs) {
|
||||
np.segs[np.count] = seg;
|
||||
np.count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- term list / object -------------------------------------------------
|
||||
|
||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||
/// the boundary rather than propagating — containment for the rare desync.
|
||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||
while (self.pos < end) {
|
||||
self.object(scope) catch break;
|
||||
}
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||
/// purpose of advancing the cursor).
|
||||
fn object(self: *Parser, scope: *Node) Error!void {
|
||||
const lead = self.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||
|
||||
_ = try self.readByte();
|
||||
switch (lead) {
|
||||
// constants and no-operand statements
|
||||
op.zero_op, op.one_op, op.ones_op => {},
|
||||
op.noop_op, op.continue_op, op.break_op, op.break_point_op => {},
|
||||
op.local0_op...op.local7_op => {},
|
||||
op.arg0_op...op.arg6_op => {},
|
||||
|
||||
// literal data
|
||||
op.byte_prefix => try self.skip(1),
|
||||
op.word_prefix => try self.skip(2),
|
||||
op.dword_prefix => try self.skip(4),
|
||||
op.qword_prefix => try self.skip(8),
|
||||
op.string_prefix => try self.skipCString(),
|
||||
|
||||
// data containers (contents skipped via their PkgLength)
|
||||
op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(),
|
||||
|
||||
// namespace modifiers / named objects
|
||||
op.name_op => try self.opName(scope),
|
||||
op.alias_op => try self.opAlias(scope),
|
||||
op.scope_op => try self.opScopeLike(scope, .scope),
|
||||
op.method_op => try self.opMethod(scope),
|
||||
op.external_op => try self.opExternal(scope),
|
||||
op.ext_op_prefix => try self.opExt(scope),
|
||||
|
||||
// control flow
|
||||
op.if_op => try self.opIf(scope),
|
||||
op.else_op => try self.opElse(scope),
|
||||
op.while_op => try self.opWhile(scope),
|
||||
op.return_op => try self.object(scope),
|
||||
op.notify_op => try self.args(scope, 2),
|
||||
|
||||
// stores / references / unary+target
|
||||
op.store_op => try self.args(scope, 2),
|
||||
op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1),
|
||||
op.increment_op, op.decrement_op => try self.args(scope, 1),
|
||||
op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2),
|
||||
op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2),
|
||||
op.copy_object_op => try self.args(scope, 2),
|
||||
|
||||
// binary + target
|
||||
op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3),
|
||||
op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3),
|
||||
op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3),
|
||||
op.divide_op => try self.args(scope, 4),
|
||||
op.to_string_op => try self.args(scope, 3),
|
||||
op.mid_op => try self.args(scope, 4),
|
||||
|
||||
// logical
|
||||
op.land_op, op.lor_op => try self.args(scope, 2),
|
||||
op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2),
|
||||
op.lnot_op => try self.opLnot(scope),
|
||||
|
||||
op.match_op => try self.opMatch(scope),
|
||||
|
||||
// CreateXField: <source> <index> NameString
|
||||
op.create_dword_field_op,
|
||||
op.create_word_field_op,
|
||||
op.create_byte_field_op,
|
||||
op.create_bit_field_op,
|
||||
op.create_qword_field_op,
|
||||
=> try self.opCreateField(scope, 2),
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse `n` operands.
|
||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) try self.object(scope);
|
||||
}
|
||||
|
||||
/// A NameString in operand/statement position: a method invocation (consuming
|
||||
/// the callee's declared argument count) or a plain name reference.
|
||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| {
|
||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||
try self.args(scope, node.arg_count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||
/// the contents are pure data, never namespace declarations.
|
||||
fn skipPkg(self: *Parser) Error!void {
|
||||
const start = self.pos;
|
||||
const len = try self.readPkgLength();
|
||||
const end = start + len;
|
||||
if (end > self.aml.len) return error.Truncated;
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
// --- namespace objects --------------------------------------------------
|
||||
|
||||
fn opName(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
const val_start = self.pos;
|
||||
try self.object(scope); // the DataRefObject value
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
||||
node.value = self.aml[val_start..self.pos];
|
||||
}
|
||||
|
||||
fn opAlias(self: *Parser, scope: *Node) Error!void {
|
||||
_ = try self.readNameString(); // source
|
||||
const np = try self.readNameString(); // the alias name
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias);
|
||||
}
|
||||
|
||||
fn opMethod(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
const flags = try self.readByte();
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method);
|
||||
node.arg_count = flags & 0x7;
|
||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||
// inside a method are created at *runtime*, not at load, so they must not
|
||||
// become permanent namespace nodes.
|
||||
node.value = self.aml[self.pos..@min(end, self.aml.len)];
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
fn opExternal(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
_ = try self.readByte(); // object type
|
||||
const arg_count = try self.readByte();
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external);
|
||||
node.arg_count = arg_count;
|
||||
}
|
||||
|
||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||
fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn opProcessor(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte)
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn opPowerRes(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||
fn opRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
const space = try self.readByte();
|
||||
const off_start = self.pos;
|
||||
try self.object(scope);
|
||||
const off_end = self.pos;
|
||||
try self.object(scope);
|
||||
const len_end = self.pos;
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
||||
node.region_space = space;
|
||||
node.region_offset_aml = self.aml[off_start..off_end];
|
||||
node.region_len_aml = self.aml[off_end..len_end];
|
||||
}
|
||||
|
||||
fn opDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
||||
}
|
||||
|
||||
fn opMutex(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
try self.skip(1); // sync flags
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex);
|
||||
}
|
||||
|
||||
fn opEvent(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event);
|
||||
}
|
||||
|
||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||
fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||
try self.args(scope, count);
|
||||
const np = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
||||
}
|
||||
|
||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||
/// Field, the first NameString is the backing region — captured so field units
|
||||
/// carry a region + bit position the evaluator can read/write.
|
||||
fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
var region: ?*Node = null;
|
||||
var i: u8 = 0;
|
||||
while (i < name_strings) : (i += 1) {
|
||||
const np = try self.readNameString();
|
||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||
if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice());
|
||||
}
|
||||
if (bank) try self.object(scope); // bank value TermArg
|
||||
const flags = try self.readByte();
|
||||
self.fieldList(end, scope, region, flags & 0x0F);
|
||||
}
|
||||
|
||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||
var bit_offset: u32 = 0;
|
||||
var access = initial_access;
|
||||
while (self.pos < end) {
|
||||
const lead = self.peek() orelse break;
|
||||
switch (lead) {
|
||||
0x00 => { // ReservedField: advances the bit position
|
||||
self.pos += 1;
|
||||
const width = self.readPkgLength() catch break;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||
self.pos += 1;
|
||||
const at = self.readByte() catch break;
|
||||
self.skip(1) catch break;
|
||||
access = at & 0x0F;
|
||||
},
|
||||
0x02 => { // ConnectField: NameString | BufferData
|
||||
self.pos += 1;
|
||||
self.object(scope) catch break;
|
||||
},
|
||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||
self.pos += 1;
|
||||
self.skip(3) catch break;
|
||||
},
|
||||
else => { // NamedField: NameSeg + PkgLength (bit width)
|
||||
const seg = self.readNameSeg() catch break;
|
||||
const width = self.readPkgLength() catch break;
|
||||
const unit = self.namespace.newFieldUnit(scope, seg) catch break;
|
||||
unit.region = region;
|
||||
unit.bit_offset = bit_offset;
|
||||
unit.bit_width = @intCast(width);
|
||||
unit.access_type = access;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
}
|
||||
}
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn opIf(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
if (self.peek() == op.else_op) {
|
||||
self.pos += 1;
|
||||
try self.opElse(scope);
|
||||
}
|
||||
}
|
||||
|
||||
fn opElse(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn opWhile(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn opLnot(self: *Parser, scope: *Node) Error!void {
|
||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||
// otherwise it is a plain LNot of one operand.
|
||||
const b = self.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => {
|
||||
self.pos += 1;
|
||||
try self.args(scope, 2);
|
||||
},
|
||||
else => try self.object(scope),
|
||||
}
|
||||
}
|
||||
|
||||
fn opMatch(self: *Parser, scope: *Node) Error!void {
|
||||
try self.object(scope); // search package
|
||||
try self.skip(1); // match opcode 1
|
||||
try self.object(scope); // operand 1
|
||||
try self.skip(1); // match opcode 2
|
||||
try self.object(scope); // operand 2
|
||||
try self.object(scope); // start index
|
||||
}
|
||||
|
||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||
|
||||
fn opExt(self: *Parser, scope: *Node) Error!void {
|
||||
const e = try self.readByte();
|
||||
switch (e) {
|
||||
op.ext.mutex => try self.opMutex(scope),
|
||||
op.ext.event => try self.opEvent(scope),
|
||||
op.ext.op_region => try self.opRegion(scope),
|
||||
op.ext.data_region => try self.opDataRegion(scope),
|
||||
op.ext.field => try self.opField(scope, 1, false),
|
||||
op.ext.index_field => try self.opField(scope, 2, false),
|
||||
op.ext.bank_field => try self.opField(scope, 2, true),
|
||||
op.ext.device => try self.opScopeLike(scope, .device),
|
||||
op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone),
|
||||
op.ext.processor => try self.opProcessor(scope),
|
||||
op.ext.power_res => try self.opPowerRes(scope),
|
||||
|
||||
op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target
|
||||
op.ext.create_field => try self.opCreateField(scope, 3),
|
||||
op.ext.load_table => try self.args(scope, 6),
|
||||
op.ext.load => try self.args(scope, 2), // NameString, Target
|
||||
op.ext.stall, op.ext.sleep => try self.args(scope, 1),
|
||||
op.ext.acquire => {
|
||||
try self.object(scope); // mutex SuperName
|
||||
try self.skip(2); // timeout WordData
|
||||
},
|
||||
op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1),
|
||||
op.ext.wait => try self.args(scope, 2),
|
||||
op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2),
|
||||
op.ext.fatal => {
|
||||
try self.skip(5); // Type(byte) + Code(dword)
|
||||
try self.object(scope); // Arg TermArg
|
||||
},
|
||||
op.ext.revision, op.ext.debug, op.ext.timer => {},
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
||||
b == op.name_char_underscore or
|
||||
b == op.root_char or
|
||||
b == op.parent_prefix_char or
|
||||
b == op.dual_name_prefix or
|
||||
b == op.multi_name_prefix;
|
||||
}
|
||||
@@ -1,93 +0,0 @@
|
||||
//! The firmware-agnostic discovery facade.
|
||||
//!
|
||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
||||
//! without ever naming ACPI or device-tree — the same way it imports `arch`
|
||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
||||
//! because a single image — a future ARM kernel especially — may boot under
|
||||
//! either firmware. That's a deliberate divergence from `arch`, which is a
|
||||
//! compile-time choice.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const device = @import("device.zig");
|
||||
const acpi = @import("acpi.zig");
|
||||
const power = @import("power.zig");
|
||||
const devicetree = @import("devicetree.zig");
|
||||
|
||||
pub const DeviceTree = device.DeviceTree;
|
||||
pub const Device = device.Device;
|
||||
pub const DeviceClass = device.DeviceClass;
|
||||
pub const Hal = device.Hal;
|
||||
pub const PowerInfo = acpi.PowerInfo;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInfo = acpi.PlatformInfo;
|
||||
pub const RegAccess = acpi.RegAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
pub fn powerInfo() PowerInfo {
|
||||
return acpi.power_info;
|
||||
}
|
||||
|
||||
/// The scalar firmware facts the arch layer needs to avoid legacy assumptions
|
||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
||||
pub fn platformInfo() PlatformInfo {
|
||||
return acpi.platform_info;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// The usable logical processors discovered during enumeration — one entry per
|
||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||
/// same instant once the application processors are started. Today only the
|
||||
/// bootstrap processor is actually running, so starting the rest is the pending SMP
|
||||
/// step (see docs/smp.md). Borrowed from static storage populated by `discover`.
|
||||
pub fn cpus() []const Cpu {
|
||||
return acpi.cpu_info.cpus[0..acpi.cpu_info.count];
|
||||
}
|
||||
|
||||
/// Non-zero only if enumeration found more processors than the static pool holds
|
||||
/// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent.
|
||||
pub fn cpusDropped() usize {
|
||||
return acpi.cpu_info.dropped;
|
||||
}
|
||||
|
||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
||||
/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for
|
||||
/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up
|
||||
/// beyond the tree's own allocations.
|
||||
pub fn discover(
|
||||
boot_info: *const danos.BootInfo,
|
||||
allocator: std.mem.Allocator,
|
||||
hal: Hal,
|
||||
) !DeviceTree {
|
||||
var dt = try DeviceTree.init(allocator);
|
||||
|
||||
if (boot_info.acpi_rsdp != 0) {
|
||||
try acpi.discover(boot_info.acpi_rsdp, &dt, hal);
|
||||
} else {
|
||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||
// path is a stub, so this reports the machine described itself no way we
|
||||
// understand yet.
|
||||
try devicetree.discover(&dt);
|
||||
}
|
||||
|
||||
return dt;
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
@@ -1,106 +0,0 @@
|
||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
||||
//!
|
||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
||||
//!
|
||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
||||
//! re-initialisation, a milestone of its own.
|
||||
|
||||
const acpi = @import("acpi.zig");
|
||||
const device = @import("device.zig");
|
||||
const Hal = device.Hal;
|
||||
|
||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||
|
||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||
pub fn enable(hal: Hal) void {
|
||||
const pi = acpi.power_info;
|
||||
if (!pi.pm1a_cnt.present()) return;
|
||||
if (readReg(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||
|
||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||
var spins: usize = 0;
|
||||
while (readReg(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
}
|
||||
|
||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
const pi = acpi.power_info;
|
||||
|
||||
// 1. The FADT reset register, when the firmware advertises support.
|
||||
if (pi.reset_supported and pi.reset.present()) {
|
||||
writeReg(hal, pi.reset, pi.reset_value);
|
||||
delay();
|
||||
}
|
||||
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYS_RST).
|
||||
hal.pioWrite(1, 0xCF9, 0x0E);
|
||||
hal.pioWrite(1, 0xCF9, 0x06);
|
||||
delay();
|
||||
// 3. Pulse the 8042 keyboard controller's reset line.
|
||||
hal.pioWrite(1, 0x64, 0xFE);
|
||||
delay();
|
||||
}
|
||||
|
||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
enable(hal);
|
||||
const pi = acpi.power_info;
|
||||
const s5 = pi.s5 orelse return;
|
||||
|
||||
if (pi.pm1a_cnt.present()) {
|
||||
writeReg(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
}
|
||||
if (pi.pm1b_cnt.present()) {
|
||||
writeReg(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
}
|
||||
delay();
|
||||
}
|
||||
|
||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
||||
_ = hal;
|
||||
return error.Unsupported;
|
||||
}
|
||||
|
||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
||||
/// [12:10], SLP_EN in bit 13.
|
||||
fn sleepValue(slp_typ: u8) u32 {
|
||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||
}
|
||||
|
||||
fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
||||
if (reg.mmio) {
|
||||
hal.mapMmio(reg.address, reg.address, true);
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
||||
return p.*;
|
||||
}
|
||||
return hal.pioRead(reg.width, @intCast(reg.address));
|
||||
}
|
||||
|
||||
fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void {
|
||||
if (reg.mmio) {
|
||||
hal.mapMmio(reg.address, reg.address, true);
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
||||
p.* = value;
|
||||
} else {
|
||||
hal.pioWrite(reg.width, @intCast(reg.address), value);
|
||||
}
|
||||
}
|
||||
|
||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||
/// the next method. The empty asm is an arch-neutral barrier that keeps the loop
|
||||
/// from being optimised away.
|
||||
fn delay() void {
|
||||
var i: usize = 0;
|
||||
while (i < 50_000_000) : (i += 1) {
|
||||
asm volatile ("" ::: .{ .memory = true });
|
||||
}
|
||||
}
|
||||
@@ -1,395 +0,0 @@
|
||||
//! Local APIC and its timer — the source of device interrupts.
|
||||
//!
|
||||
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
|
||||
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
|
||||
//! timer, which is the simplest device interrupt to bring up: it needs no
|
||||
//! external routing, just a vector and a count. We use it as danos's heartbeat.
|
||||
//!
|
||||
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
|
||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
||||
//! the LAPIC won't deliver the next one.
|
||||
|
||||
const io = @import("io.zig");
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference: an I/O port or MMIO counter.
|
||||
pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool };
|
||||
|
||||
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
|
||||
// the legacy-safe assumptions so the code still works if discovery never ran.
|
||||
var cfg_pic_present: bool = true;
|
||||
var cfg_hpet_base: u64 = 0; // 0 = no HPET discovered
|
||||
var cfg_pm_timer: ?PmTimer = null;
|
||||
/// Which reference the last calibration used, for logging.
|
||||
var cal_source: []const u8 = "none";
|
||||
|
||||
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
|
||||
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
|
||||
cfg_pic_present = pic_present;
|
||||
cfg_hpet_base = hpet_base;
|
||||
cfg_pm_timer = pm_timer;
|
||||
}
|
||||
|
||||
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
|
||||
pub fn calibrationSource() []const u8 {
|
||||
return cal_source;
|
||||
}
|
||||
|
||||
/// IDT vector the timer fires on (in the device range, >= 32).
|
||||
pub const timer_vector = 32;
|
||||
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
|
||||
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
|
||||
const spurious_vector = 47;
|
||||
|
||||
// LAPIC register offsets.
|
||||
const reg_spurious = 0x0F0;
|
||||
const reg_eoi = 0x0B0;
|
||||
const reg_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
|
||||
const reg_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
|
||||
const reg_lvt_timer = 0x320;
|
||||
const reg_timer_initial = 0x380;
|
||||
const reg_timer_current = 0x390;
|
||||
const reg_timer_divide = 0x3E0;
|
||||
|
||||
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
|
||||
|
||||
const lvt_masked = 1 << 16;
|
||||
const lvt_periodic = 1 << 17;
|
||||
const timer_divide_16 = 0x3;
|
||||
|
||||
const ia32_apic_base_msr = 0x1B;
|
||||
|
||||
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
|
||||
/// the MSR and so register writes compile to normal stores rather than a
|
||||
/// `mov moffs`, which the self-hosted backend can't encode.
|
||||
var base: usize = 0xFEE00000;
|
||||
|
||||
var tick_count: u64 = 0;
|
||||
|
||||
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
|
||||
/// At divide-by-16, this is the effective counting rate.
|
||||
var ticks_per_ms: u32 = 0;
|
||||
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
|
||||
var timer_hz: u32 = 0;
|
||||
|
||||
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
|
||||
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
|
||||
/// monotonic clock — far finer than the millisecond timer tick.
|
||||
var tsc_hz: u64 = 0;
|
||||
var tsc_base: u64 = 0;
|
||||
|
||||
/// Read the 64-bit Time Stamp Counter.
|
||||
fn rdtsc() u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdtsc"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
fn read(reg: u32) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(base + reg)).*;
|
||||
}
|
||||
fn write(reg: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + reg)).* = value;
|
||||
}
|
||||
|
||||
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
||||
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
|
||||
fn remapAndMaskPic() void {
|
||||
io.outb(0x20, 0x11); // start init (cascade mode)
|
||||
io.outb(0xA0, 0x11);
|
||||
io.outb(0x21, 0x20); // master offset 0x20
|
||||
io.outb(0xA1, 0x28); // slave offset 0x28
|
||||
io.outb(0x21, 0x04); // tell master about slave on IRQ2
|
||||
io.outb(0xA1, 0x02);
|
||||
io.outb(0x21, 0x01); // 8086 mode
|
||||
io.outb(0xA1, 0x01);
|
||||
io.outb(0x21, 0xFF); // mask all
|
||||
io.outb(0xA1, 0xFF);
|
||||
}
|
||||
|
||||
/// Enable the Local APIC: mask the PIC (only if one is present — a legacy-free
|
||||
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
|
||||
/// software-enable the APIC via its spurious-vector register.
|
||||
pub fn init() void {
|
||||
if (cfg_pic_present) remapAndMaskPic();
|
||||
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
|
||||
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||
}
|
||||
|
||||
/// Software-enable *this* core's Local APIC — the application-processor counterpart
|
||||
/// of `init`, minus the one-time PIC remap (the BSP already masked it) and minus
|
||||
/// calibration (the timer rate is a shared hardware constant, measured once). Each
|
||||
/// core has its own LAPIC at the same MMIO address, so no per-core base is needed.
|
||||
pub fn initSecondary() void {
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
write(reg_spurious, 0x100 | spurious_vector); // software enable
|
||||
}
|
||||
|
||||
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
|
||||
|
||||
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
|
||||
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
|
||||
pub fn sendInit(apic_id: u32) void {
|
||||
write(reg_icr_high, apic_id << 24);
|
||||
write(reg_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
|
||||
waitIcrIdle();
|
||||
}
|
||||
|
||||
/// Send a STARTUP IPI (SIPI) telling the target core to begin executing at physical
|
||||
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
|
||||
/// sent twice after the INIT; both calls block until delivery completes.
|
||||
pub fn sendStartup(apic_id: u32, vector: u8) void {
|
||||
write(reg_icr_high, apic_id << 24);
|
||||
write(reg_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
|
||||
waitIcrIdle();
|
||||
}
|
||||
|
||||
fn waitIcrIdle() void {
|
||||
while (read(reg_icr_low) & icr_delivery_pending != 0) {}
|
||||
}
|
||||
|
||||
/// The calibration window: we time everything against a 10 ms reference interval.
|
||||
const calib_ms = 10;
|
||||
|
||||
/// Measure the LAPIC timer's and the TSC's rates. The PIT (legacy 8254) can be
|
||||
/// absent on UEFI Class 3 firmware — and polling it would hang — so we pick a
|
||||
/// reference clock in order of preference: the CPU's own TSC frequency (CPUID leaf
|
||||
/// 0x15, no external timer needed), then the discovered HPET, then the ACPI PM
|
||||
/// timer, and only the PIT as a last resort. Each path yields the same two rates.
|
||||
pub fn calibrate() void {
|
||||
var done = false;
|
||||
|
||||
// 1. CPUID leaf 0x15 gives the TSC frequency directly — measure the LAPIC
|
||||
// against the TSC itself, needing no external timer at all.
|
||||
if (cpuidTscHz()) |hz| {
|
||||
measure(hz, ~@as(u64, 0), rdtsc);
|
||||
tsc_hz = hz; // keep the exact enumerated value
|
||||
cal_source = "cpuid";
|
||||
done = true;
|
||||
}
|
||||
|
||||
// 2. The discovered HPET.
|
||||
if (!done and cfg_hpet_base != 0) {
|
||||
if (hpetHz()) |hpet_hz| {
|
||||
measure(hpet_hz, hpetMask(), readHpet);
|
||||
cal_source = "hpet";
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 3. The ACPI PM timer (fixed 3.579545 MHz).
|
||||
if (!done) {
|
||||
if (cfg_pm_timer) |pt| {
|
||||
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
|
||||
cal_source = "pm-timer";
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. The legacy PIT, last resort.
|
||||
if (!done) {
|
||||
calibratePit();
|
||||
cal_source = "pit";
|
||||
}
|
||||
|
||||
// A bad measurement (no reference actually ticked) leaves nonsense; fall back.
|
||||
if (ticks_per_ms == 0 or tsc_hz == 0) {
|
||||
calibratePit();
|
||||
cal_source = "pit";
|
||||
}
|
||||
|
||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||
}
|
||||
|
||||
/// Run the LAPIC timer one-shot from its max count while a monotonic reference
|
||||
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
|
||||
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
|
||||
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
|
||||
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
|
||||
|
||||
write(reg_timer_divide, timer_divide_16);
|
||||
write(reg_lvt_timer, lvt_masked);
|
||||
write(reg_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
const ref0 = refNow();
|
||||
const tsc0 = rdtsc();
|
||||
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
|
||||
const tsc1 = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
||||
write(reg_timer_initial, 0);
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
|
||||
}
|
||||
|
||||
/// The PIT fallback (legacy 8254 channel 2, polled). Only reached when no better
|
||||
/// reference exists — on a legacy-free machine this path isn't taken.
|
||||
fn calibratePit() void {
|
||||
const pit_hz = 1_193_182;
|
||||
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
||||
|
||||
write(reg_timer_divide, timer_divide_16);
|
||||
write(reg_lvt_timer, lvt_masked);
|
||||
write(reg_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
||||
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
||||
io.outb(0x42, @truncate(pit_count));
|
||||
io.outb(0x42, @truncate(pit_count >> 8));
|
||||
|
||||
const tsc_start = rdtsc();
|
||||
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
|
||||
var guard: u64 = 0;
|
||||
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
|
||||
const tsc_end = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
||||
write(reg_timer_initial, 0);
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
|
||||
}
|
||||
|
||||
// --- reference clocks ------------------------------------------------------
|
||||
|
||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
||||
fn cpuidTscHz() ?u64 {
|
||||
if (cpuid(0).eax < 0x15) return null;
|
||||
const r = cpuid(0x15);
|
||||
if (r.eax == 0 or r.ebx == 0 or r.ecx == 0) return null; // ratio/crystal not given
|
||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||
}
|
||||
|
||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||
|
||||
fn cpuid(leaf: u32) CpuidRegs {
|
||||
var a: u32 = undefined;
|
||||
var b: u32 = undefined;
|
||||
var c: u32 = undefined;
|
||||
var d: u32 = undefined;
|
||||
asm volatile ("cpuid"
|
||||
: [a] "={eax}" (a),
|
||||
[b] "={ebx}" (b),
|
||||
[c] "={ecx}" (c),
|
||||
[d] "={edx}" (d),
|
||||
: [leaf] "{eax}" (leaf),
|
||||
[sub] "{ecx}" (@as(u32, 0)),
|
||||
);
|
||||
return .{ .eax = a, .ebx = b, .ecx = c, .edx = d };
|
||||
}
|
||||
|
||||
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
|
||||
// 64-bit-counter capable), general config at +0x10, main counter at +0xF0.
|
||||
fn hpetRead64(off: usize) u64 {
|
||||
return @as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).*;
|
||||
}
|
||||
fn hpetWrite64(off: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).* = value;
|
||||
}
|
||||
|
||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||
fn hpetHz() ?u64 {
|
||||
paging.map(cfg_hpet_base & ~@as(u64, 0xFFF), cfg_hpet_base & ~@as(u64, 0xFFF), true);
|
||||
const caps = hpetRead64(0x00);
|
||||
const period_fs = caps >> 32; // femtoseconds per tick
|
||||
if (period_fs == 0) return null;
|
||||
hpetWrite64(0x10, hpetRead64(0x10) | 1); // ENABLE_CNF: start the main counter
|
||||
return 1_000_000_000_000_000 / period_fs; // 1e15 fs/s ÷ fs/tick
|
||||
}
|
||||
|
||||
/// The HPET counter width mask (64- or 32-bit, per caps bit 13).
|
||||
fn hpetMask() u64 {
|
||||
return if (hpetRead64(0x00) & (1 << 13) != 0) ~@as(u64, 0) else 0xFFFF_FFFF;
|
||||
}
|
||||
|
||||
fn readHpet() u64 {
|
||||
return hpetRead64(0xF0);
|
||||
}
|
||||
|
||||
fn readPmTimer() u64 {
|
||||
const pt = cfg_pm_timer.?;
|
||||
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(pt.address)).*;
|
||||
return io.inl(@intCast(pt.address));
|
||||
}
|
||||
|
||||
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
|
||||
/// calibrate() to have run.
|
||||
pub fn initTimer(hz: u32) void {
|
||||
timer_hz = hz;
|
||||
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
||||
write(reg_timer_divide, timer_divide_16);
|
||||
write(reg_lvt_timer, timer_vector | lvt_periodic);
|
||||
write(reg_timer_initial, @intCast(count));
|
||||
}
|
||||
|
||||
/// Configured periodic-interrupt frequency (Hz).
|
||||
pub fn frequencyHz() u32 {
|
||||
return timer_hz;
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
|
||||
pub fn lapicHz() u64 {
|
||||
return @as(u64, ticks_per_ms) * 1000;
|
||||
}
|
||||
|
||||
/// Measured TSC frequency (Hz).
|
||||
pub fn tscHz() u64 {
|
||||
return tsc_hz;
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
||||
// the scheduler uses for sleep deadlines.
|
||||
|
||||
pub fn nanos() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn micros() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn millis() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
||||
}
|
||||
|
||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||
pub fn eoi() void {
|
||||
write(reg_eoi, 0);
|
||||
}
|
||||
|
||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
||||
var on_tick: ?*const fn () void = null;
|
||||
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
on_tick = hook;
|
||||
}
|
||||
|
||||
/// The timer interrupt handler: advance the monotonic tick count, then run the
|
||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
||||
pub fn timerTick() void {
|
||||
tick_count +%= 1;
|
||||
if (on_tick) |hook| hook();
|
||||
}
|
||||
|
||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
||||
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
||||
pub fn ticks() u64 {
|
||||
return @as(*const volatile u64, &tick_count).*;
|
||||
}
|
||||
@@ -1,351 +0,0 @@
|
||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
||||
//! architecture is added by pointing that module at a different directory in
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const serial = @import("serial.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const ioapic = @import("ioapic.zig");
|
||||
const io = @import("io.zig");
|
||||
const smp = @import("smp.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||
/// so it can be the very first thing called.
|
||||
pub fn serialInit() void {
|
||||
serial.init();
|
||||
}
|
||||
|
||||
/// Write bytes to the serial port.
|
||||
pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Emit a one-byte checkpoint to the POST diagnostic port (0x80). A POST card or
|
||||
/// BMC displays it; it's the last-resort progress signal when there's no text
|
||||
/// output at all. Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
||||
pub fn postCode(code: u8) void {
|
||||
io.outb(0x80, code);
|
||||
}
|
||||
|
||||
/// Whether a Bochs/QEMU-style debug console is on port 0xE9 (it returns 0xE9 when
|
||||
/// read). On real hardware the port reads back 0xFF, so this stays false — a safe
|
||||
/// probe before we write to it.
|
||||
pub fn debugconPresent() bool {
|
||||
return io.inb(0xE9) == 0xE9;
|
||||
}
|
||||
|
||||
/// Output sink: write bytes to the 0xE9 debug console (see `debugconPresent`).
|
||||
pub fn debugconWrite(bytes: []const u8) void {
|
||||
for (bytes) |b| io.outb(0xE9, b);
|
||||
}
|
||||
|
||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
||||
/// (setFaultHandler) first so early faults are caught.
|
||||
pub fn init() void {
|
||||
gdt.init();
|
||||
tss.init();
|
||||
idt.init();
|
||||
}
|
||||
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
paging.init(allocFrame, boot_info);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
||||
paging.map(virt, phys, writable);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
pub fn unmapPage(virt: u64) void {
|
||||
paging.unmap(virt);
|
||||
}
|
||||
|
||||
/// CR3 holds the physical address of the active top-level page table.
|
||||
pub fn readCr3() u64 {
|
||||
return asm volatile ("mov %%cr3, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
|
||||
/// data pointer (there's no user mode yet, so no `swapgs` dance — GS base is always
|
||||
/// the running core's per-CPU block). Set once per core during bring-up, after the
|
||||
/// GDT is loaded (loading a GS *selector* would otherwise clobber this base).
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
|
||||
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
|
||||
/// core calls this once, after its GDT is in place.
|
||||
pub fn setCpuLocal(ptr: usize) void {
|
||||
io.wrmsr(ia32_gs_base, ptr);
|
||||
}
|
||||
|
||||
/// This core's per-CPU data pointer (the value `setCpuLocal` stored). Reads the GS
|
||||
/// base MSR — a per-core register, so each core sees its own without any locking.
|
||||
pub fn cpuLocal() usize {
|
||||
return io.rdmsr(ia32_gs_base);
|
||||
}
|
||||
|
||||
// --- SMP: application-processor bring-up ----------------------------------
|
||||
|
||||
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
||||
/// The frame stays inert (zeroed, non-executable) between wakes and is armed only
|
||||
/// while a core is climbing — so a core can be (re)woken at any time (retry, or a
|
||||
/// future power manager) without leaving an executable page resident. See smp.zig.
|
||||
pub fn setTrampolinePage(phys: u64) void {
|
||||
smp.setTrampolinePage(phys);
|
||||
}
|
||||
|
||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it
|
||||
/// `stack_top` and its per-CPU pointer `percpu`; it adopts the current (kernel) page
|
||||
/// tables. Returns false if it doesn't come online within the timeout. Blocks until
|
||||
/// the core reports in.
|
||||
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
|
||||
return smp.startAp(apic_id, stack_top, percpu, index, readCr3());
|
||||
}
|
||||
|
||||
/// Register the generic entry a woken AP jumps to once its arch state is up (its own
|
||||
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
smp.setSecondaryEntry(entry);
|
||||
}
|
||||
|
||||
/// Test hook: force the next `n` AP wake attempts to fail, so the retry path can be
|
||||
/// exercised deterministically (see the smp-retry test). No effect when `n` is 0.
|
||||
pub fn testFailNextWakes(n: u32) void {
|
||||
smp.testFailNextWakes(n);
|
||||
}
|
||||
|
||||
/// The reserved AP-trampoline frame (0 if none). For tests that check it's inert.
|
||||
pub fn trampolinePage() u64 {
|
||||
return smp.trampolinePage();
|
||||
}
|
||||
|
||||
/// Whether the page at `virt` is currently mapped executable (present, NX clear).
|
||||
pub fn pageExecutable(virt: u64) bool {
|
||||
return paging.isExecutable(virt);
|
||||
}
|
||||
|
||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
||||
pub const timer_hz = 1000;
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference (re-exported for the config).
|
||||
pub const PmTimer = apic.PmTimer;
|
||||
/// A MADT interrupt-source override (re-exported for the config).
|
||||
pub const IsoEntry = ioapic.IsoEntry;
|
||||
|
||||
/// Discovered platform facts the arch layer needs so it makes no legacy
|
||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
||||
pub const PlatformConfig = struct {
|
||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
||||
pic_present: bool = true,
|
||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
||||
hpet_base: u64 = 0,
|
||||
/// The ACPI PM timer, another calibration reference.
|
||||
pm_timer: ?PmTimer = null,
|
||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
||||
ioapic_base: u64 = 0,
|
||||
ioapic_gsi_base: u32 = 0,
|
||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
||||
overrides: []const IsoEntry = &.{},
|
||||
};
|
||||
|
||||
/// Apply the discovered platform config. Must run before `startTimer` (the timer
|
||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
||||
/// Maps + masks the I/O APIC immediately.
|
||||
pub fn configurePlatform(cfg: PlatformConfig) void {
|
||||
apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer);
|
||||
ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides);
|
||||
ioapic.init();
|
||||
}
|
||||
|
||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
||||
pub fn serialReconfigure(is_mmio: bool, addr: u64) void {
|
||||
serial.reconfigure(is_mmio, addr);
|
||||
}
|
||||
|
||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
||||
pub fn timerCalibrationSource() []const u8 {
|
||||
return apic.calibrationSource();
|
||||
}
|
||||
|
||||
/// I/O APIC diagnostics (for boot logging / verification).
|
||||
pub fn ioapicEntryCount() u32 {
|
||||
return ioapic.entryCount();
|
||||
}
|
||||
pub fn ioapicEntryLow(n: u32) u32 {
|
||||
return ioapic.entryLow(n);
|
||||
}
|
||||
|
||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
||||
pub fn startTimer() void {
|
||||
apic.init();
|
||||
apic.calibrate();
|
||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
||||
apic.initTimer(timer_hz);
|
||||
}
|
||||
|
||||
/// Number of timer ticks since startTimer().
|
||||
pub fn ticks() u64 {
|
||||
return apic.ticks();
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
||||
pub fn nanos() u64 {
|
||||
return apic.nanos();
|
||||
}
|
||||
pub fn micros() u64 {
|
||||
return apic.micros();
|
||||
}
|
||||
pub fn millis() u64 {
|
||||
return apic.millis();
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
||||
pub fn lapicHz() u64 {
|
||||
return apic.lapicHz();
|
||||
}
|
||||
pub fn tscHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
}
|
||||
|
||||
/// Mask maskable interrupts (`cli`).
|
||||
pub fn disableInterrupts() void {
|
||||
asm volatile ("cli");
|
||||
}
|
||||
|
||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
||||
/// restoreInterrupts.
|
||||
pub fn saveInterrupts() u64 {
|
||||
var flags: u64 = undefined;
|
||||
asm volatile (
|
||||
\\pushfq
|
||||
\\pop %[f]
|
||||
\\cli
|
||||
: [f] "=r" (flags),
|
||||
:
|
||||
: .{ .memory = true }
|
||||
);
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
||||
pub fn restoreInterrupts(flags: u64) void {
|
||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
||||
}
|
||||
|
||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
apic.setTickHook(hook);
|
||||
}
|
||||
|
||||
// --- context switching (for the scheduler) -------------------------------
|
||||
|
||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||
|
||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
||||
switch_context(old_rsp, new_rsp);
|
||||
}
|
||||
|
||||
/// Build the initial stack for a new task so that switching to it lands in
|
||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
||||
/// The layout must match switch_context's push order (callee-saved, then the
|
||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
||||
var sp = stack_top;
|
||||
const push = struct {
|
||||
fn f(p: *usize, value: usize) void {
|
||||
p.* -= @sizeOf(usize);
|
||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
||||
}
|
||||
}.f;
|
||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
||||
push(&sp, 0); // rbx
|
||||
push(&sp, 0); // rbp
|
||||
push(&sp, 0); // r12
|
||||
push(&sp, 0); // r13
|
||||
push(&sp, 0); // r14
|
||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
||||
return sp;
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
idt.on_fault = handler;
|
||||
}
|
||||
|
||||
/// A human-readable name for a CPU exception vector.
|
||||
pub fn vectorName(vector: u64) []const u8 {
|
||||
return idt.vectorName(vector);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
||||
/// MMIO-only architecture this would be implemented differently.
|
||||
pub fn pioRead(width: u8, port: u16) u32 {
|
||||
return switch (width) {
|
||||
1 => io.inb(port),
|
||||
2 => io.inw(port),
|
||||
4 => io.inl(port),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
||||
switch (width) {
|
||||
1 => io.outb(port, @truncate(value)),
|
||||
2 => io.outw(port, @truncate(value)),
|
||||
4 => io.outl(port, value),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
||||
pub fn readCr2() u64 {
|
||||
return asm volatile ("mov %%cr2, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
||||
/// docs/halting.md for the full reasoning.
|
||||
pub fn halt() noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Spin-wait hint (`pause`). Emitted in the body of a spinlock's busy-wait: it
|
||||
/// relaxes the core while it polls a contended lock — yielding pipeline resources
|
||||
/// to a hyperthread sibling and easing the cache-coherency traffic on the lock
|
||||
/// line. Purely a performance/power hint; correct to omit, but kinder on the bus.
|
||||
pub fn cpuRelax() void {
|
||||
asm volatile ("pause");
|
||||
}
|
||||
@@ -1,97 +0,0 @@
|
||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
||||
//! ACPI's MADT (via discovery), never assumed.
|
||||
//!
|
||||
//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own
|
||||
//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now
|
||||
//! is `init`, which maps the I/O APIC and **masks every input**, the correct
|
||||
//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real
|
||||
//! device driver (a keyboard, say).
|
||||
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
||||
|
||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
||||
var gsi_base: u32 = 0;
|
||||
var max_entries: u32 = 0;
|
||||
var overrides: [16]IsoEntry = undefined;
|
||||
var override_count: usize = 0;
|
||||
|
||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
||||
const reg_ioregsel = 0x00;
|
||||
const reg_iowin = 0x10;
|
||||
const reg_version = 0x01;
|
||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
||||
|
||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
||||
base = ioapic_base;
|
||||
gsi_base = ioapic_gsi_base;
|
||||
override_count = @min(isos.len, overrides.len);
|
||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
||||
}
|
||||
|
||||
fn regRead(index: u32) u32 {
|
||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
||||
return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*;
|
||||
}
|
||||
fn regWrite(index: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
||||
@as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value;
|
||||
}
|
||||
|
||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
||||
regWrite(redir_base + 2 * n, low);
|
||||
regWrite(redir_base + 2 * n + 1, high);
|
||||
}
|
||||
|
||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
||||
pub fn init() void {
|
||||
if (base == 0) return;
|
||||
paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true);
|
||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
||||
var n: u32 = 0;
|
||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
||||
}
|
||||
|
||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
||||
/// first device driver.
|
||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
||||
if (base == 0) return;
|
||||
|
||||
var gsi: u32 = irq;
|
||||
var flags: u16 = 0;
|
||||
for (overrides[0..override_count]) |o| {
|
||||
if (o.source == irq) {
|
||||
gsi = o.gsi;
|
||||
flags = o.flags;
|
||||
}
|
||||
}
|
||||
if (gsi < gsi_base) return;
|
||||
const n = gsi - gsi_base;
|
||||
if (n >= max_entries) return;
|
||||
|
||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
||||
var low: u32 = vector;
|
||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
||||
writeEntry(n, low, high);
|
||||
}
|
||||
|
||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
||||
pub fn entryCount() u32 {
|
||||
return max_entries;
|
||||
}
|
||||
|
||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
||||
pub fn entryLow(n: u32) u32 {
|
||||
if (base == 0) return 0;
|
||||
return regRead(redir_base + 2 * n);
|
||||
}
|
||||
@@ -1,185 +0,0 @@
|
||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
||||
#
|
||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
||||
|
||||
.text
|
||||
|
||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
||||
# set with mov, so we far-return through the caller's own return address.
|
||||
.global gdt_flush
|
||||
gdt_flush:
|
||||
lgdt (%rdi)
|
||||
mov $0x10, %ax # kernel data selector
|
||||
mov %ax, %ds
|
||||
mov %ax, %es
|
||||
mov %ax, %ss
|
||||
mov %ax, %fs
|
||||
mov %ax, %gs
|
||||
pop %rax # caller's return address
|
||||
push $0x08 # kernel code selector (new CS)
|
||||
push %rax # return address (new RIP)
|
||||
lretq
|
||||
|
||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
||||
.global idt_flush
|
||||
idt_flush:
|
||||
lidt (%rdi)
|
||||
ret
|
||||
|
||||
# load_tr(di = TSS selector): load the task register.
|
||||
.global load_tr
|
||||
load_tr:
|
||||
ltr %di
|
||||
ret
|
||||
|
||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
||||
# Cooperative context switch: save the callee-saved registers on the current
|
||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
||||
# registers are the compiler's responsibility (this looks like a normal call).
|
||||
.global switch_context
|
||||
switch_context:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
||||
mov %rsi, %rsp # switch to the new task's stack
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret # return into the new task's saved instruction pointer
|
||||
|
||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
||||
# leaves its entry function in r15. A fresh task is switched to with the big kernel
|
||||
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
|
||||
# so it releases the lock here before running its body. r15 survives the call (it's
|
||||
# callee-saved). New tasks then start with interrupts enabled.
|
||||
.extern releaseForFreshTask
|
||||
.global task_trampoline
|
||||
task_trampoline:
|
||||
call releaseForFreshTask # drop the kernel lock we inherited across the switch
|
||||
sti
|
||||
call *%r15 # call the task entry (fn() void)
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $0
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
||||
.macro STUB_ERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
STUB_NOERR 0
|
||||
STUB_NOERR 1
|
||||
STUB_NOERR 2
|
||||
STUB_NOERR 3
|
||||
STUB_NOERR 4
|
||||
STUB_NOERR 5
|
||||
STUB_NOERR 6
|
||||
STUB_NOERR 7
|
||||
STUB_ERR 8
|
||||
STUB_NOERR 9
|
||||
STUB_ERR 10
|
||||
STUB_ERR 11
|
||||
STUB_ERR 12
|
||||
STUB_ERR 13
|
||||
STUB_ERR 14
|
||||
STUB_NOERR 15
|
||||
STUB_NOERR 16
|
||||
STUB_ERR 17
|
||||
STUB_NOERR 18
|
||||
STUB_NOERR 19
|
||||
STUB_NOERR 20
|
||||
STUB_ERR 21
|
||||
STUB_NOERR 22
|
||||
STUB_NOERR 23
|
||||
STUB_NOERR 24
|
||||
STUB_NOERR 25
|
||||
STUB_NOERR 26
|
||||
STUB_NOERR 27
|
||||
STUB_NOERR 28
|
||||
STUB_NOERR 29
|
||||
STUB_NOERR 30
|
||||
STUB_NOERR 31
|
||||
|
||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
||||
# code, so they all use the dummy-zero form.
|
||||
STUB_NOERR 32
|
||||
STUB_NOERR 33
|
||||
STUB_NOERR 34
|
||||
STUB_NOERR 35
|
||||
STUB_NOERR 36
|
||||
STUB_NOERR 37
|
||||
STUB_NOERR 38
|
||||
STUB_NOERR 39
|
||||
STUB_NOERR 40
|
||||
STUB_NOERR 41
|
||||
STUB_NOERR 42
|
||||
STUB_NOERR 43
|
||||
STUB_NOERR 44
|
||||
STUB_NOERR 45
|
||||
STUB_NOERR 46
|
||||
STUB_NOERR 47
|
||||
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
isr_common:
|
||||
push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop the vector and error code
|
||||
iretq
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user